Compare commits
23
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
56fd779f7a | ||
|
|
c9c8b1d06c | ||
|
|
7278c354f0 | ||
|
|
8032ea682b | ||
|
|
948fb03630 | ||
|
|
7ff0dbb6e1 | ||
|
|
bccc14ffef | ||
|
|
33f8607e17 | ||
|
|
2eeba3df6f | ||
|
|
cb5dda4f45 | ||
|
|
4847882e46 | ||
|
|
33cf3dd29a | ||
|
|
9abf81535a | ||
|
|
2dd1194f33 | ||
|
|
705c8729ca | ||
|
|
01ff3bb5f0 | ||
|
|
d63d3cc7fc | ||
|
|
c887f80f57 | ||
|
|
c9bf22d483 | ||
|
|
c737490a53 | ||
|
|
09a39ddf37 | ||
|
|
29a65016d0 | ||
|
|
7ac4f6b723 |
+6
-3
@@ -1,9 +1,12 @@
|
||||
.git
|
||||
# .git is sent without its config. Without a VERSION build argument the
|
||||
# stage that compiles runs `git describe --tags --always` on .git, which
|
||||
# does not need .git/config; that file can hold a credential, such as a
|
||||
# password in a remote URL or the token the CI checkout step stores there.
|
||||
.git/config
|
||||
|
||||
.claude
|
||||
.DS_Store
|
||||
sfdupes
|
||||
files.dat
|
||||
node_modules
|
||||
*.log
|
||||
*.out
|
||||
*.test
|
||||
|
||||
@@ -6,4 +6,6 @@ jobs:
|
||||
steps:
|
||||
# actions/checkout v4.2.2, 2026-02-22
|
||||
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- run: script/cibuild
|
||||
|
||||
@@ -27,7 +27,6 @@ node_modules/
|
||||
*.log
|
||||
|
||||
# Local scan data
|
||||
files.dat
|
||||
*.sqlite
|
||||
*.sqlite-shm
|
||||
*.sqlite-wal
|
||||
|
||||
+96
-38
@@ -12,9 +12,9 @@ COPY . .
|
||||
# build would exit 0 having run nothing. script/cibuild and
|
||||
# script/docker pass a fresh CHECK_EPOCH on every invocation.
|
||||
#
|
||||
# Two properties this depends on. ARG is per-stage, so the build stage
|
||||
# below declares it again; one declaration here would leave that
|
||||
# stage's gate cacheable. And each gate RUN must reference the value,
|
||||
# Two properties this depends on. ARG is per-stage, so the markdown and
|
||||
# build stages below declare it again; one declaration here would leave
|
||||
# their gates cacheable. And each gate RUN must reference the value,
|
||||
# because BuildKit hashes the expanded command: a declared but
|
||||
# unreferenced ARG invalidates nothing.
|
||||
#
|
||||
@@ -27,12 +27,16 @@ ARG CHECK_EPOCH
|
||||
# target now runs `docker build -f Dockerfile.lint`, and a docker build
|
||||
# cannot run a docker build: routing the gate through make would mean
|
||||
# nesting docker inside this image. Same reason `make check` is gone
|
||||
# from the build stage below.
|
||||
#
|
||||
# `make fmt-check` is not run in this stage: it now also runs prettier
|
||||
# over Markdown, and this golangci-lint image has no node. The gate runs
|
||||
# in the build stage below, where script/bootstrap installs node and
|
||||
# prettier.
|
||||
# from the build stage below, and `make fmt-check` from both stages: it
|
||||
# runs prettier through docker too. Its gofmt half is the step below,
|
||||
# its Markdown half the markdown stage further down. gofmt's output is
|
||||
# assigned to a variable first so that its own exit status, as when it
|
||||
# cannot parse a file, still fails the step.
|
||||
RUN echo "gate gofmt, epoch ${CHECK_EPOCH}" && \
|
||||
files="$(gofmt -s -l .)" && \
|
||||
if [ -n "$files" ]; then \
|
||||
echo "gofmt: files not formatted:" >&2; echo "$files" >&2; exit 1; \
|
||||
fi
|
||||
|
||||
# The FROM above and the one in Dockerfile.lint pin the same linter
|
||||
# twice, and nothing else keeps them in sync; this fails the build when
|
||||
@@ -49,71 +53,125 @@ RUN echo "gate config verify, epoch ${CHECK_EPOCH}" && \
|
||||
RUN echo "gate lint, epoch ${CHECK_EPOCH}" && \
|
||||
golangci-lint run --config .golangci.yml ./...
|
||||
|
||||
# Prettier stage: the prettier that formats this repository's Markdown,
|
||||
# never installed on a host. script/fmt and script/fmt-check build this
|
||||
# stage alone and run it with the repository mounted on /src. prettier
|
||||
# is installed in /tools so that the repository, mounted or copied onto
|
||||
# /src, cannot hide it.
|
||||
# node:22-alpine, 2026-02-22
|
||||
FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS prettier
|
||||
WORKDIR /tools
|
||||
# yarn.lock pins prettier by hash, and --frozen-lockfile fails rather
|
||||
# than install anything yarn.lock does not name.
|
||||
COPY package.json yarn.lock ./
|
||||
RUN yarn install --frozen-lockfile
|
||||
ENV PATH=/tools/node_modules/.bin:$PATH
|
||||
WORKDIR /src
|
||||
|
||||
# Markdown stage: the Markdown half of `make fmt-check`, as a gate.
|
||||
FROM prettier AS markdown
|
||||
COPY . .
|
||||
# Second per-stage declaration of the gate cache-buster; see the lint
|
||||
# stage above.
|
||||
ARG CHECK_EPOCH
|
||||
RUN echo "gate prettier, epoch ${CHECK_EPOCH}" && \
|
||||
prettier --check '**/*.md' --tab-width 4 --prose-wrap always
|
||||
|
||||
# Build stage
|
||||
# golang:1.25-alpine, 2026-07-23
|
||||
FROM golang@sha256:56961d79ea8129efddcc0b8643fd8a5416b4e6228cfd477e3fd61deb2672c587 AS builder
|
||||
|
||||
# We never build or run as root. Create an unprivileged user and point
|
||||
# HOME and the Go caches at its home so go build and go test can write
|
||||
# their caches when we drop to it below. $GOPATH/bin is deliberately not
|
||||
# on PATH: script/bootstrap no longer `go install`s anything (the linter
|
||||
# runs from a pinned image, never from a host install), so nothing lands
|
||||
# HOME and the build cache at its home so go build and go test can write
|
||||
# it when we drop to it below. $GOPATH/bin is deliberately not on PATH:
|
||||
# script/bootstrap no longer `go install`s anything (the linter runs
|
||||
# from a pinned image, never from a host install), so nothing lands
|
||||
# there and adding it would only widen what this image resolves.
|
||||
#
|
||||
# The module cache is kept outside that home, at the base image's
|
||||
# default /go/pkg/mod, and belongs to root: script/bootstrap fills it as
|
||||
# root. Do not move it into the home and hand it over with `chown -R`:
|
||||
# that walks every file in it, which took from about 80 s to over ten
|
||||
# minutes on a shared host, depending on load.
|
||||
RUN adduser -D -u 1000 builder
|
||||
ENV HOME=/home/builder
|
||||
ENV GOPATH=/home/builder/go
|
||||
ENV GOMODCACHE=/go/pkg/mod
|
||||
ENV GOCACHE=/home/builder/.cache/go-build
|
||||
|
||||
WORKDIR /src
|
||||
|
||||
# No-op file copy whose only purpose is the build-graph edge: it is what
|
||||
# makes this stage depend on the lint stage, and so what forces BuildKit
|
||||
# to finish fmt-check, the pin guard and lint before compilation and
|
||||
# tests start. Remove it and the fail-fast design dies silently — the
|
||||
# build stops gating on lint and still exits 0. It replaces a copy of
|
||||
# the linter binary itself, which is no longer wanted here: nothing in
|
||||
# this stage runs the linter, because `make lint` is now a docker build
|
||||
# and a docker build cannot run inside one.
|
||||
# No-op file copies whose only purpose is the build-graph edge: they are
|
||||
# what make this stage depend on the lint and markdown stages, and so
|
||||
# what forces BuildKit to finish gofmt, the pin guard, lint and prettier
|
||||
# before compilation and tests start. Remove one and the fail-fast
|
||||
# design dies silently — the build stops gating on that stage and still
|
||||
# exits 0. The first replaces a copy of the linter binary itself, which
|
||||
# is no longer wanted here: nothing in this stage runs the linter,
|
||||
# because `make lint` is now a docker build and a docker build cannot
|
||||
# run inside one.
|
||||
COPY --from=lint /src/go.sum /dev/null
|
||||
COPY --from=markdown /src/go.sum /dev/null
|
||||
|
||||
# Install development prerequisites the same way a developer does,
|
||||
# rather than duplicating the installs inline. Only script/ and the
|
||||
# dependency manifests are copied first, nothing else, so this layer
|
||||
# stays cached until the scripts or the dependencies change — bootstrap
|
||||
# runs `go mod download` and `yarn install`, which is why there is no
|
||||
# separate invocation of either here. The JS manifests (package.json,
|
||||
# yarn.lock) are copied too so the yarn install layer caches alongside
|
||||
# the Go one.
|
||||
# ends in `go mod download`, which is why there is no separate
|
||||
# invocation of it here.
|
||||
COPY script/ script/
|
||||
COPY go.mod go.sum package.json yarn.lock ./
|
||||
COPY go.mod go.sum ./
|
||||
RUN script/bootstrap
|
||||
|
||||
COPY . .
|
||||
# Hand builder only what it writes to, without walking the module cache.
|
||||
# This layer stays cached with bootstrap.
|
||||
# - /src itself: make build writes the binary into it, and git refuses
|
||||
# a repository whose top directory belongs to another user.
|
||||
# - the module cache's cache/download directory itself, not what is in
|
||||
# it: Go only reads the downloaded modules, but make build saves its
|
||||
# lookup of this module's own version from git there, in a new
|
||||
# directory named after the module path.
|
||||
# - builder's home: the go commands bootstrap ran as root left Go's
|
||||
# telemetry files there, a few small files.
|
||||
RUN chown builder:builder /src /go/pkg/mod/cache/download && \
|
||||
chown -R builder:builder /home/builder
|
||||
|
||||
# Hand the sources and caches to the unprivileged user, then drop root
|
||||
# before running any checks or builds.
|
||||
RUN chown -R builder:builder /src /home/builder
|
||||
# The sources are handed to builder as they are copied, so no layer has
|
||||
# to walk them. Then drop root before running any checks or builds.
|
||||
COPY --chown=builder:builder . .
|
||||
USER builder
|
||||
|
||||
# Fail the build unless the branch is green. Runs as non-root so the
|
||||
# permission-denied test paths are exercised legitimately (root would
|
||||
# bypass the chmod(0) the tests rely on).
|
||||
#
|
||||
# The gates are the individual targets, not `make check`: that aggregate
|
||||
# runs `script/lint`, which is now a docker build, and nothing inside an
|
||||
# image build may shell out to docker. Lint is not skipped by this — it
|
||||
# ran in the lint stage above, which this stage's COPY --from makes a
|
||||
# prerequisite. `make`, not the scripts directly, because the Makefile's
|
||||
# The gate is `make test`, not `make check`: that aggregate runs
|
||||
# `script/lint` and `script/fmt-check`, which both run docker, and
|
||||
# nothing inside an image build may shell out to docker. Lint and the
|
||||
# format checks are not skipped by this — they ran in the lint and
|
||||
# markdown stages above, which this stage's COPY --from lines make
|
||||
# prerequisites. `make`, not the script directly, because the Makefile's
|
||||
# `export CGO_ENABLED = 0` applies only to what it invokes.
|
||||
#
|
||||
# Second per-stage declaration of the gate cache-buster; see the lint
|
||||
# Third per-stage declaration of the gate cache-buster; see the lint
|
||||
# stage above for why one is not enough. It is placed after USER so the
|
||||
# drop to the unprivileged user still happens before the checks run.
|
||||
ARG CHECK_EPOCH
|
||||
RUN echo "gate test, epoch ${CHECK_EPOCH}" && make test
|
||||
RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
|
||||
|
||||
RUN make build
|
||||
# The version stamped into the binary: the VERSION build argument when
|
||||
# one is given, otherwise `git describe --tags --always` of the .git in
|
||||
# the build context (git is installed by script/bootstrap above). A
|
||||
# context that carries .git and still yields no version fails the build;
|
||||
# with neither, as from a source tarball, it is "dev".
|
||||
ARG VERSION
|
||||
RUN version="${VERSION:-$(git describe --tags --always || echo dev)}"; \
|
||||
if [ -e .git ] && { [ -z "$version" ] || [ "$version" = dev ] || \
|
||||
[ "$version" = unknown ]; }; then \
|
||||
echo "no version could be derived although the build context carries .git" >&2; \
|
||||
exit 1; \
|
||||
fi; \
|
||||
make build VERSION="$version"
|
||||
|
||||
# Runtime stage
|
||||
# alpine:3.22, 2026-07-23
|
||||
|
||||
@@ -46,4 +46,4 @@ hooks:
|
||||
@script/install-precommit
|
||||
|
||||
clean:
|
||||
rm -f $(BINARY) files.dat
|
||||
rm -f $(BINARY)
|
||||
|
||||
@@ -4,16 +4,21 @@
|
||||
|
||||
`sfdupes` is an MIT-licensed Go CLI tool by [@sneak](https://sneak.berlin) that
|
||||
quickly identifies _candidate_ duplicate files — and, ultimately, entire
|
||||
duplicate directory trees — across very large filesystems without reading full
|
||||
file contents. Files are considered duplicates when they have identical size,
|
||||
identical SHA-256 of their first 1024 bytes, and identical SHA-256 of their last
|
||||
1024 bytes. This is a strong candidate signal, not proof of identical content
|
||||
(the middle of the file is never read); the intended use is finding duplicate
|
||||
downloads and duplicated directory trees on multi-terabyte ZFS servers where
|
||||
reading every byte is prohibitively expensive. `scan` maintains a persistent
|
||||
SQLite database of file signatures that survives between runs, so it can be run
|
||||
from cron and the reports can be generated at any time from the most recent
|
||||
scan.
|
||||
duplicate directory trees — across very large filesystems without reading every
|
||||
byte of every file. Files are considered duplicates when their sizes are equal
|
||||
and they agree on a short ladder of hashes. A file under 10 MiB is hashed in
|
||||
full and compared directly. A larger file is gated first on the SHA-256 of its
|
||||
first 64 KiB and of its last 64 KiB, and only when its size and both of those
|
||||
match another file's is it read for a content hash to compare — the SHA-256 of
|
||||
the whole file when it is under 50 MiB, or of gigabyte-spaced 1 MiB samples when
|
||||
it is 50 MiB or larger. Below 50 MiB the content hash is proof of identical
|
||||
content; at or above 50 MiB it is a strong candidate signal rather than proof,
|
||||
because the gaps between samples are never read. The intended use is finding
|
||||
duplicate downloads and duplicated directory trees on multi-terabyte ZFS servers
|
||||
where reading every byte of every file is prohibitively expensive. `scan`
|
||||
maintains a persistent SQLite database of file signatures that survives between
|
||||
runs, so it can be run from cron and the reports can be generated at any time
|
||||
from the most recent scan.
|
||||
|
||||
This README is the complete and authoritative specification.
|
||||
|
||||
@@ -28,8 +33,9 @@ export SFDUPES_DATABASE="$HOME/.local/share/sfdupes/db.sqlite"
|
||||
```
|
||||
|
||||
`scan` walks one or more filesystem trees and maintains one database record per
|
||||
regular file (path, size, mtime, head hash, tail hash). The database persists
|
||||
between runs; a rescan only hashes files that are new or changed, and removes
|
||||
regular file (path, size, mtime, head hash, tail hash, content hash). The
|
||||
database persists between runs; a rescan only hashes files that are new or
|
||||
changed, or that may have gained a duplicate since the last scan, and removes
|
||||
records for files that no longer exist. `report` reads the database and prints
|
||||
the file-level duplicates report. `trees` reads the same database and prints the
|
||||
duplicate-tree report. A missing/invalid subcommand — or a `scan` invocation
|
||||
@@ -40,18 +46,134 @@ by setting `SFDUPES_DATABASE`. The intended deployment is a daily `sfdupes scan`
|
||||
cron job, with the reporting commands run interactively whenever needed; their
|
||||
results are as fresh as the last completed scan.
|
||||
|
||||
### Install
|
||||
|
||||
With Go installed, this builds and installs the current `main` branch:
|
||||
|
||||
```sh
|
||||
go install sneak.berlin/go/sfdupes@main
|
||||
```
|
||||
|
||||
The binary goes to `$(go env GOPATH)/bin`, or to `$GOBIN` when that is set. A
|
||||
binary installed this way reports its version as `dev`; one built from a clone
|
||||
or into the Docker image carries the git tag or commit it was built from.
|
||||
|
||||
From a clone, `make build` writes the binary to `./sfdupes`:
|
||||
|
||||
```sh
|
||||
git clone https://git.eeqj.de/sneak/sfdupes.git
|
||||
cd sfdupes
|
||||
make build
|
||||
```
|
||||
|
||||
Copy the binary to `/usr/local/bin` for the cron job below.
|
||||
|
||||
`make docker` builds the Docker image, tagged `sfdupes`, after running the tests
|
||||
and the linter (see "Build"). The image runs `sfdupes` as root with the database
|
||||
at its default path, so a bind mount of `/var/lib/sfdupes` keeps the database
|
||||
between runs. Mount the scanned tree at the same path inside the container as on
|
||||
the host; read-only is enough. The database records paths as the container sees
|
||||
them, so the reports then name the host's paths.
|
||||
|
||||
```sh
|
||||
make docker
|
||||
docker run --rm -v /srv:/srv:ro -v /var/lib/sfdupes:/var/lib/sfdupes \
|
||||
sfdupes scan /srv
|
||||
docker run --rm -v /var/lib/sfdupes:/var/lib/sfdupes sfdupes report > dupes.tsv
|
||||
```
|
||||
|
||||
### Daily scan from cron
|
||||
|
||||
Run `scan` as root, so that it can read every file: a path it cannot read is
|
||||
skipped with a warning and loses its database record (see "Rules for the walk").
|
||||
As a file `/etc/cron.d/sfdupes`:
|
||||
|
||||
```
|
||||
30 3 * * * root /usr/local/bin/sfdupes scan /srv 2>>/var/log/sfdupes.log || tail -n 3 /var/log/sfdupes.log
|
||||
```
|
||||
|
||||
- The database is `/var/lib/sfdupes/db.sqlite`, created with its directory by
|
||||
the first scan. To keep it elsewhere, set
|
||||
`SFDUPES_DATABASE=/path/to/db.sqlite` before the command on the same line.
|
||||
- `scan` writes nothing to stdout. Its stderr, appended here to
|
||||
`/var/log/sfdupes.log`, holds a plain progress line as each phase starts and
|
||||
then at most every 5 seconds, a warning for each path it skips, and the
|
||||
summary line (see "Progress" and "`scan` mode"). The log grows with every
|
||||
scan; rotate it like any other.
|
||||
- Skipped paths do not fail a scan: it still exits 0, and cron sends nothing. A
|
||||
scan that fails, or is stopped by `SIGINT` or `SIGTERM`, exits 1 with the
|
||||
reason among the last lines of the log; `tail` prints them, and cron mails
|
||||
them to root if the host can send mail.
|
||||
- A scan still running when the next one starts carries on. The new one fails at
|
||||
once, and the lines cron mails include
|
||||
`sfdupes: another scan is running (lock held on /var/lib/sfdupes/db.sqlite.lock)`.
|
||||
- `report` and `trees` need only read access to the database (see "Database").
|
||||
Under the usual umask of `022` the first scan creates it readable by every
|
||||
user, so an unprivileged user can run them against root's database.
|
||||
|
||||
### Reading the reports
|
||||
|
||||
Each row of `report` names two copies of one file, and each row of `trees` two
|
||||
copies of one directory tree (see "Report output format" and "Trees output
|
||||
format"). In a group of copies, the path that sorts first byte by byte is
|
||||
`first` and every other path is a `dupe` of it. `first` says nothing about which
|
||||
copy is the original or the oldest; which copy to keep is your choice.
|
||||
|
||||
A row is a candidate, not proof:
|
||||
|
||||
- The reports read only the database, so they show the files as of the last
|
||||
scan; a file may have changed or gone since.
|
||||
- A file of 50 MiB or more is compared only on samples of its content (see
|
||||
"Duplicate detection").
|
||||
- Paths that are hard links to one file are listed as duplicates, but they share
|
||||
their data, so removing one frees nothing.
|
||||
|
||||
Compare a pair byte for byte before removing either copy. For the row
|
||||
`/srv/a/big.iso`, `/srv/b/big-copy.iso`, `4294967296`:
|
||||
|
||||
```sh
|
||||
cmp /srv/a/big.iso /srv/b/big-copy.iso && echo identical
|
||||
[ /srv/a/big.iso -ef /srv/b/big-copy.iso ] && echo "hard links"
|
||||
```
|
||||
|
||||
`cmp` prints nothing and exits 0 only when every byte matches, and otherwise
|
||||
reports where the files differ. The second line prints `hard links` when the two
|
||||
paths are the same file, so removing either frees nothing.
|
||||
|
||||
A path holding a backslash, tab, newline or carriage return is escaped in the
|
||||
reports (see "Report output format"). Undo the escapes before using it.
|
||||
`printf '%b'` does exactly that, because every backslash in an escaped path
|
||||
starts one of the four escapes. Command substitution drops trailing newlines, so
|
||||
print an `x` after the path and remove it afterwards, or a path that ends in a
|
||||
newline names a different file:
|
||||
|
||||
```sh
|
||||
p="$(printf '%bx' '/srv/a/tab\tname.txt')"; p="${p%x}"
|
||||
cmp "$p" /srv/b/tab-copy.txt
|
||||
```
|
||||
|
||||
Check a `trees` row with `diff -r`, which compares the two trees file by file
|
||||
and also names anything present in only one of them, such as an empty directory
|
||||
or a symlink, which `trees` does not see.
|
||||
|
||||
## Rationale
|
||||
|
||||
Duplicate finders that hash entire files do not scale to the target environment:
|
||||
~10 million files and ~150 TB on possibly slow or busy disks (a ZFS pool under
|
||||
resilver). Reading at most 2 KiB per file — and only from files whose size at
|
||||
least one other file shares, since a size-unique file cannot be a duplicate —
|
||||
makes a full-filesystem sweep tractable, and the signatures are kept in a
|
||||
persistent database, so the expensive filesystem pass is incremental: a rescan
|
||||
re-hashes only files whose recorded mtime or size changed, and all analysis
|
||||
happens offline from the database alone. The end goal is not individual files
|
||||
but whole duplicated trees — duplicate extractions, duplicate downloads, copied
|
||||
project trees — which an operator can consider removing as a unit.
|
||||
resilver). sfdupes spends disk I/O only on files whose size at least one other
|
||||
file shares, since a size-unique file cannot be a duplicate. Of those, a file
|
||||
under 10 MiB is read in full; a larger one has its cheap end windows read first,
|
||||
and is read for a content hash only when its size and both end windows match
|
||||
another file's — the whole file below 50 MiB, but only gigabyte-spaced samples
|
||||
at or above 50 MiB, so the largest files are never read in full. This keeps a
|
||||
full-filesystem sweep tractable, and the signatures are kept in a persistent
|
||||
database, so the expensive filesystem pass is incremental: a rescan re-hashes
|
||||
only files whose recorded mtime or size changed, plus — for its content hash — a
|
||||
file of 10 MiB or more whose size and end windows have come to match another
|
||||
file's. All analysis happens offline from the database alone. The end goal is
|
||||
not individual files but whole duplicated trees — duplicate extractions,
|
||||
duplicate downloads, copied project trees — which an operator can consider
|
||||
removing as a unit.
|
||||
|
||||
## Design
|
||||
|
||||
@@ -63,27 +185,43 @@ Goals, in order:
|
||||
trees), so the operator can consider removing an entire subtree at once.
|
||||
File-level duplicate detection is the foundation; tree-level detection is
|
||||
built on top of it.
|
||||
2. **Never read full file contents.** At most 2 KiB is read per file (first and
|
||||
last 1024 bytes), and only files whose size at least one other file shares
|
||||
are read at all — a size-unique file cannot be a duplicate. Scale target:
|
||||
tens of millions of files, ~150 TB filesystem, possibly slow or busy disks
|
||||
(ZFS pool under resilver). Holding one small record (path, size, mtime) per
|
||||
file in memory during a scan is acceptable; holding every file's hashes is
|
||||
not (they stay in the database).
|
||||
2. **Spend I/O in proportion to duplicate likelihood.** Only files whose size
|
||||
at least one other file shares are read at all — a size-unique file cannot
|
||||
be a duplicate. Those are compared by the ladder in "Duplicate detection"
|
||||
below: a file under 10 MiB is hashed in full, while a larger file is gated
|
||||
on cheap 64 KiB end windows first, and gets a content hash only when its
|
||||
size and both end windows match another file's. That hash reads the whole
|
||||
file below 50 MiB but only gigabyte-spaced 1 MiB samples at or above it, so
|
||||
the very largest files are still never read in full. Scale target: tens of
|
||||
millions of files, ~150 TB filesystem, possibly slow or busy disks (ZFS pool
|
||||
under resilver). Holding one small record (path, size, mtime) per file in
|
||||
memory during a scan is acceptable; holding every file's hashes is not (they
|
||||
stay in the database). The reporting commands do not hold every file's
|
||||
hashes either: `report` lets SQLite group and order the records and writes
|
||||
each row as it reads it, so its memory does not grow with the database, and
|
||||
`trees` reads the records in path order and keeps each directory's path,
|
||||
digest and totals, plus the hashes of only the files in the directories
|
||||
holding the record being read, so its memory grows with the number of
|
||||
directories and with the size of the largest directory.
|
||||
3. **Scan incrementally, analyze offline.** The expensive filesystem scan
|
||||
maintains a persistent database; an unchanged file is never read again on a
|
||||
rescan. All analysis (`report`, `trees`) works from the database alone and
|
||||
must never touch the scanned filesystem again. `scan` is designed to be
|
||||
cronned; the reports run at any time against the last completed scan.
|
||||
rescan, except to compute its content hash once a file of 10 MiB or more
|
||||
comes to match another on size and both end windows. All analysis (`report`,
|
||||
`trees`) works from the database alone and must never touch the scanned
|
||||
filesystem again. `scan` is designed to be cronned; the reports run at any
|
||||
time against the last completed scan.
|
||||
4. **Clean stream separation.** Everything on stdout is machine-readable data.
|
||||
All progress, warnings, and summaries go to stderr. Never mix them.
|
||||
All progress, warnings, summaries, and help and usage text go to stderr.
|
||||
Never mix them.
|
||||
|
||||
### Constraints
|
||||
|
||||
- Language: Go (module `sneak.berlin/go/sfdupes`). Binary name: `sfdupes`.
|
||||
- Dependencies: standard library, `github.com/spf13/cobra` for the CLI, **one
|
||||
progress-bar library** (`github.com/schollz/progressbar/v3`), and **one SQLite
|
||||
driver** (`modernc.org/sqlite`, pure Go, so builds keep cgo disabled).
|
||||
progress-bar library** (`github.com/schollz/progressbar/v3`),
|
||||
`golang.org/x/term` to tell whether stderr is a terminal, **one SQLite
|
||||
driver** (`modernc.org/sqlite`, pure Go, so builds keep cgo disabled), and
|
||||
`golang.org/x/sys` for `flock(2)` (the scan lock, see "Database").
|
||||
`github.com/spf13/viper` is permitted if configuration-file support is ever
|
||||
needed, but is not currently used. No other third-party deps.
|
||||
- Cross-compilation is not a concern. Builds run with cgo disabled (the
|
||||
@@ -107,8 +245,18 @@ Three subcommands, all implemented:
|
||||
sfdupes scan [--workers N] [-x] PATH...
|
||||
sfdupes report > dupes.tsv
|
||||
sfdupes trees > dupetrees.tsv
|
||||
sfdupes --version
|
||||
sfdupes [command] --help
|
||||
```
|
||||
|
||||
`--workers N` sets the size of each `scan` worker pool (default: the number of
|
||||
CPUs), and `-x` (`--one-file-system`) keeps the walk of each operand on that
|
||||
operand's filesystem; both are described under "`scan` mode".
|
||||
`sfdupes --version` (or `-v`) prints one line, `sfdupes VERSION`, to stdout and
|
||||
exits 0, writing nothing to stderr. `-h` or `--help`, alone or after a
|
||||
subcommand, prints the help text to stderr and exits 0, writing nothing to
|
||||
stdout.
|
||||
|
||||
### Database
|
||||
|
||||
All three subcommands operate on a single SQLite database file:
|
||||
@@ -119,33 +267,112 @@ All three subcommands operate on a single SQLite database file:
|
||||
- `scan` creates the database (and its parent directory) on first use. `report`
|
||||
and `trees` require an existing database; a missing database file is a fatal
|
||||
error (exit 1) telling the user to run `scan` first.
|
||||
- The database uses WAL journal mode and a busy timeout, so running a report
|
||||
while a cron `scan` is in progress is safe. The filesystem is authoritative;
|
||||
the database is an eventually-consistent reflection of it. Hashed records are
|
||||
committed in batched transactions while the scan is still running (keeping the
|
||||
WAL small and letting concurrent reports observe progress), so a report may
|
||||
see a scan's changes partially applied, and a scan that dies partway leaves a
|
||||
valid database holding everything hashed so far; the next scan skips those
|
||||
records and converges toward the filesystem.
|
||||
- Only one `scan` runs against a database at a time. For its whole run, `scan`
|
||||
holds an exclusive `flock(2)` lock on a lock file beside the database, named
|
||||
by appending `.lock` to the database path (`/var/lib/sfdupes/db.sqlite.lock`
|
||||
by default), taken before it walks the filesystem or opens the database. A
|
||||
second `scan` against the same database does not wait: it fails at once with a
|
||||
one-line error naming the lock file and exits 1, without walking anything or
|
||||
opening the database, and the running scan carries on. The lock file is
|
||||
created on first use, open to its owner only, and left in place: a leftover
|
||||
file blocks nothing, because the lock ends with the process holding it however
|
||||
it ends, a fatal error or an interrupt included, and deleting the file while a
|
||||
scan runs would let a second scan start. `report` and `trees` never take the
|
||||
lock, so they run during a scan.
|
||||
- While `scan` runs, the database is in WAL journal mode with a busy timeout, so
|
||||
running a report while a cron `scan` is in progress is safe. The filesystem is
|
||||
authoritative; the database is an eventually-consistent reflection of it.
|
||||
Hashed records are committed in batched transactions while the scan is still
|
||||
running (keeping the WAL small and letting concurrent reports observe
|
||||
progress), so a report may see a scan's changes partially applied, and a scan
|
||||
that dies partway leaves a valid database holding every batch committed so far
|
||||
(an interrupted scan also commits the batch in progress, see "Error handling
|
||||
and exit codes"); the next scan skips those records and converges toward the
|
||||
filesystem.
|
||||
- `scan` switches the database back to rollback-journal mode when it closes it,
|
||||
so between scans the database file alone holds the whole database. Each switch
|
||||
needs the database to itself: a `scan` that starts while a report is still
|
||||
reading waits for it up to the 10-second busy timeout, then fails; a `scan`
|
||||
that ends while a report has the database open warns and leaves the database
|
||||
in WAL mode until the next scan. `report` writes each row as it reads it, so
|
||||
it is still reading while its output is paused (a pager, a stalled pipe), and
|
||||
a `scan` started then fails after the busy timeout.
|
||||
- `report` and `trees` open the database read-only and need only read access to
|
||||
the database file, and no write access to its directory. While the database is
|
||||
in WAL mode they also read the `-wal` and `-shm` files beside it, which SQLite
|
||||
creates with the database file's permissions.
|
||||
- Schema (`PRAGMA user_version` is the schema version, currently 1; a database
|
||||
with any other version is a fatal error):
|
||||
with any other version is a fatal error. `scan` creates the schema and sets
|
||||
the version in one transaction, so a first scan stopped while doing so leaves
|
||||
an empty database the next scan sets up. A database at version 0 that already
|
||||
has a `files` table was therefore not made by sfdupes; every subcommand
|
||||
refuses it with an error telling the user to remove the file and rescan):
|
||||
|
||||
```sql
|
||||
CREATE TABLE files (
|
||||
path BLOB PRIMARY KEY, -- absolute path, raw bytes
|
||||
size INTEGER NOT NULL, -- bytes, from lstat
|
||||
mtime INTEGER NOT NULL, -- Unix seconds, from lstat
|
||||
head TEXT NOT NULL, -- lowercase-hex SHA-256, first 1 KiB
|
||||
tail TEXT NOT NULL -- lowercase-hex SHA-256, last 1 KiB
|
||||
head TEXT NOT NULL, -- lowercase-hex SHA-256; first 64 KiB, or whole file under 10 MiB
|
||||
tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB
|
||||
content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples
|
||||
) WITHOUT ROWID;
|
||||
CREATE INDEX files_signature ON files (size, head, tail, content);
|
||||
```
|
||||
|
||||
Paths are stored as BLOBs because Unix paths are raw bytes, not guaranteed
|
||||
UTF-8. `mtime` is used only for change detection; it is not part of the
|
||||
duplicate key. `head` and `tail` are empty strings when the file has never
|
||||
been hashed because its size was unique as of the last scan that covered it;
|
||||
such records still define the file for tree reconstruction but never
|
||||
participate in duplicate groups.
|
||||
duplicate key. For a file under 10 MiB `head`, `tail`, and `content` all
|
||||
hold the whole-file hash (that range is hashed in full, with no end
|
||||
windows); for a larger file `head` and `tail` hold the first- and last-64
|
||||
KiB hashes and `content` the whole-file or sampled hash. All three are empty
|
||||
strings when the file has never been hashed because its size was unique as
|
||||
of the last scan that covered it. For a file of 10 MiB or more, `content`
|
||||
stays empty until the content phase of a scan (see "`scan` mode" below) has
|
||||
read the file. A record with an empty `content` is never part of a duplicate
|
||||
group, though it still defines the file for tree reconstruction. The
|
||||
`files_signature` index lets SQLite group the records by signature for
|
||||
`report` without sorting the whole table.
|
||||
|
||||
### Duplicate detection
|
||||
|
||||
Two files are duplicates only when they agree on every rung of this ladder; a
|
||||
mismatch at any rung means they are not duplicates. `scan` stores each file's
|
||||
hashes, and `report` and `trees` group files by the whole signature — size,
|
||||
`head`, `tail`, and `content` — so the grouping is exactly this ladder applied
|
||||
across everything scanned into the database, even across separate scans.
|
||||
|
||||
1. **Size.** Files of different sizes are never compared. Only files whose size
|
||||
at least one other file shares are hashed at all.
|
||||
2. **Under 10 MiB: whole file.** A file smaller than 10 MiB is hashed in full
|
||||
and compared directly, with no separate end-window step — small files are
|
||||
cheap to read to the last byte, and doing so makes the comparison exact.
|
||||
`head`, `tail`, and `content` all hold this whole-file SHA-256, so such a
|
||||
file's signature is decided entirely by its size and its content.
|
||||
3. **10 MiB and above: head and tail.** For a larger file, the SHA-256 of the
|
||||
first 64 KiB (`head`) and of the last 64 KiB (`tail`) are a cheap gate that
|
||||
eliminates most same-size pairs before any bulk reading: the content hash of
|
||||
the next two rungs is computed only for a file whose size, `head`, and
|
||||
`tail` match another file's, whether that file is scanned in the same run or
|
||||
stored by an earlier scan. A stored file that first gains such a match in a
|
||||
later scan gets its content hash then; until it has one, its `content` is
|
||||
empty and it is not a duplicate. At 10 MiB and above the two windows never
|
||||
overlap.
|
||||
4. **10 MiB and above, content below 50 MiB.** The SHA-256 of the entire file.
|
||||
Agreement here is proof of identical content (barring a SHA-256 collision).
|
||||
5. **10 MiB and above, content 50 MiB and above.** A sampled SHA-256: the 1 MiB
|
||||
window at each gigabyte-aligned offset (0, 1 GiB, 2 GiB, … while inside the
|
||||
file, the final window truncated at end of file) is fed, in order, into one
|
||||
hash. This is **deliberately probabilistic** — the gaps between samples are
|
||||
never read, so two large files that agree on every sample are reported as
|
||||
duplicates without being read in full. It is the price of never reading a
|
||||
150 GB file end to end. Because size is already part of the signature, only
|
||||
equal-size files reach this rung, so their sample boundaries always align.
|
||||
|
||||
`head`, `tail`, and `content` are one column each. A file below 10 MiB and one
|
||||
at or above it never share a size, and neither do a file below 50 MiB and one at
|
||||
or above it, so a stored value is never ambiguous between the whole-file,
|
||||
end-window, and sampled forms.
|
||||
|
||||
### `scan` mode
|
||||
|
||||
@@ -161,14 +388,25 @@ pool. Overlapping operands are harmless — an operand that duplicates another o
|
||||
lies under another is dropped before walking, so every file is reached exactly
|
||||
once and produces one database record.
|
||||
|
||||
An operand that is a symlink (never followed, not even as an operand), socket,
|
||||
FIFO, or device node, or a directory named `.zfs`, is not scanned. `scan` prints
|
||||
a one-line warning naming the path and what it is, counts it as skipped, and
|
||||
drops it from the scanned operands before reading the database. Another operand
|
||||
beneath it is still scanned. The records stored beneath it are not deleted: they
|
||||
are treated like any other record outside the scanned operands, including the
|
||||
content-phase exception below. If it lies under another operand, they are under
|
||||
that operand instead, and are deleted like any other record there that this scan
|
||||
did not verify. This is not an error: a scan whose every operand is dropped
|
||||
walks nothing and exits 0.
|
||||
|
||||
`scan` synchronizes the database with the filesystem state under the scanned
|
||||
operands:
|
||||
|
||||
- Only a file whose size at least one other file shares is ever read: a
|
||||
size-unique file cannot be a duplicate, so it is recorded without hashes
|
||||
(`head` and `tail` empty). The size census covers every file walked this scan
|
||||
plus every database record outside the scanned operands, so a possible
|
||||
duplicate of a separately scanned tree is still recognized.
|
||||
(`head`, `tail`, and `content` empty). The size census covers every file
|
||||
walked this scan plus every database record outside the scanned operands, so a
|
||||
possible duplicate of a separately scanned tree is still recognized.
|
||||
- A file not yet in the database is inserted: hashed when its size is shared,
|
||||
without hashes otherwise.
|
||||
- A file already in the database is **skipped without reading its contents**
|
||||
@@ -176,7 +414,9 @@ operands:
|
||||
than the recorded mtime. This is what makes a daily rescan cheap. Exception:
|
||||
an unchanged file whose record lacks hashes is hashed — and its record updated
|
||||
— once its size becomes shared, so hashing deferred by size-uniqueness happens
|
||||
as soon as it could matter.
|
||||
as soon as it could matter. Likewise, an unchanged file of 10 MiB or more
|
||||
whose record has no `content` hash is read for one by the content phase below
|
||||
once its size, `head`, and `tail` match another record's.
|
||||
- A file whose mtime is newer than recorded, or whose size differs, is processed
|
||||
as if new: re-hashed, or recorded without hashes, per the shared-size rule.
|
||||
- A database record whose path lies under one of the scanned operands but was
|
||||
@@ -184,11 +424,16 @@ operands:
|
||||
deleted files. It also removes records for paths that failed to stat or hash
|
||||
this run: the database only ever contains signatures verified by the most
|
||||
recent scan that covered them (a subsequent successful scan re-adds such
|
||||
files).
|
||||
files). A failure in the content phase below removes nothing: the record is
|
||||
left as it is.
|
||||
- Database records outside the scanned operands are untouched, so disjoint trees
|
||||
can be scanned on different schedules into the same database.
|
||||
can be scanned on different schedules into the same database. The one
|
||||
exception is the content phase below: a stored file of 10 MiB or more without
|
||||
a `content` hash is read for one, wherever it lies, once its size, `head`, and
|
||||
`tail` match another record's. If that file is gone or has changed since its
|
||||
record was written, the record is left as it is.
|
||||
|
||||
`scan` runs **three sequential phases over the whole scan**. Parallelism lives
|
||||
`scan` runs **four sequential phases over the whole scan**. Parallelism lives
|
||||
inside each phase; batched database writes begin during the hash phase:
|
||||
|
||||
1. **walk + stat** — enumerate the trees under all `PATH` operands concurrently
|
||||
@@ -204,10 +449,11 @@ inside each phase; batched database writes begin during the hash phase:
|
||||
2. **hash** — with the census complete, each carried file's size decides its
|
||||
fate. Size-unique files are never read: new or changed ones are recorded
|
||||
without hashes in the update phase, unchanged unhashed ones simply keep
|
||||
their records. Every file with a shared size is hashed by the worker pool:
|
||||
read the first `min(1024, size)` bytes and the last `min(1024, size)` bytes
|
||||
(one read when `size <= 1024`, since the two windows coincide) and compute
|
||||
the SHA-256 of each. Zero-length files have constant hashes and are never
|
||||
their records. Every file with a shared size is hashed by the worker pool as
|
||||
described in "Duplicate detection" above: a file under 10 MiB in full, which
|
||||
gives its `head`, `tail`, and `content` alike, and a larger file only in its
|
||||
end windows, which give its `head` and `tail`; its content hash is left to
|
||||
the content phase. Zero-length files have constant hashes and are never
|
||||
opened. Files are hashed in **inode order** (minimizing seeks on spinning
|
||||
disks), and paths that are hard links to the same inode are **read once**,
|
||||
all sharing the one result — a hard-link backup farm costs one read per
|
||||
@@ -219,13 +465,36 @@ inside each phase; batched database writes begin during the hash phase:
|
||||
3. **update** — commit the final partial batch, the hash-less records for
|
||||
size-unique new and changed files, and the deletions for records the scan
|
||||
did not verify (vanished files, plus paths that failed to stat or hash).
|
||||
4. **content** — find every record of 10 MiB or more without a `content` hash
|
||||
whose size, `head`, and `tail` equal another record's, anywhere in the
|
||||
database: records from this scan and records stored by earlier scans, inside
|
||||
or outside the scanned operands. SQLite finds them, so only the records to
|
||||
be read are kept in memory, never every file's hashes. Every record sharing
|
||||
their size, `head`, and `tail`, including one that already has a `content`
|
||||
hash, has its file checked with `lstat` first. A file that is gone, is no
|
||||
longer a regular file, or has changed (a different size, or an mtime newer
|
||||
than recorded) keeps its record as it is and does not count as a match for
|
||||
the others. Any other `lstat` error is warned about and counted as skipped,
|
||||
with the same result. If such a record has no `content` hash, it stays out
|
||||
of duplicate groups; if it has one, it is still reported until a scan
|
||||
covering its own tree updates or removes it. The files that pass and have no
|
||||
`content` hash are read only if at least two of those records pass, so a
|
||||
file whose only matches are stale costs no read; a file that already has a
|
||||
`content` hash is never read again. They are read by a worker pool as in the
|
||||
hash phase, in inode order and once per inode, and their content hashes are
|
||||
committed in batches. A failed read is warned about and counted as skipped;
|
||||
its record keeps an empty `content`, so it is not a duplicate, and a later
|
||||
scan tries again.
|
||||
|
||||
Rules for the walk:
|
||||
|
||||
- Only regular files. Skip directories, symlinks (do not follow, including
|
||||
symlink operands), sockets, FIFOs, and device nodes.
|
||||
symlink operands), sockets, FIFOs, and device nodes. An operand that is a
|
||||
symlink, socket, FIFO, or device node is dropped as described in "`scan` mode"
|
||||
above.
|
||||
- Never descend into a directory named `.zfs` (ZFS snapshot pseudo-dirs; walking
|
||||
them would list every file once per snapshot).
|
||||
them would list every file once per snapshot), not even when it is an operand;
|
||||
such an operand is dropped the same way.
|
||||
- Filesystem boundaries are crossed by default. With `-x` (long form
|
||||
`--one-file-system`, following the GNU `du`/`rsync` convention), never descend
|
||||
into a directory on a different filesystem than its `PATH` operand; each
|
||||
@@ -234,17 +503,20 @@ Rules for the walk:
|
||||
unreadable): print a one-line warning to stderr, skip the path, and continue.
|
||||
Per-file errors never abort the run; the final summary reports how many were
|
||||
skipped. As specified above, a skipped path that has a database record from an
|
||||
earlier scan loses that record; an unreadable directory subtree likewise loses
|
||||
its records (accepted: the database mirrors what the latest scan could
|
||||
actually verify).
|
||||
earlier scan loses that record, unless it failed only in the content phase, or
|
||||
is an operand dropped before the database was read that lies under no other
|
||||
operand; an unreadable directory subtree likewise loses its records (accepted:
|
||||
the database mirrors what the latest scan could actually verify).
|
||||
|
||||
Concurrency: the walk phase (which also stats files) and the hash phase each use
|
||||
a worker pool of `--workers` workers (default `runtime.NumCPU()`); the walk
|
||||
parallelizes across directories, hashing across files. Both phases are
|
||||
seek-bound on spinning disks, so raising `--workers` well past the core count
|
||||
can help on pools with many spindles. The main goroutine owns partitioning,
|
||||
database writes, and progress rendering; progress display must never block the
|
||||
workers.
|
||||
Concurrency: the walk phase (which also stats files), the hash phase, and the
|
||||
content phase each use a worker pool of `--workers` workers (default
|
||||
`runtime.NumCPU()`); the walk parallelizes across directories, hashing across
|
||||
files. `--workers` must be at least 1: a smaller value is a usage error,
|
||||
reported in one line on stderr with exit 2 before anything is scanned. All three
|
||||
phases are seek-bound on spinning disks, so raising `--workers` well past the
|
||||
core count can help on pools with many spindles. The main goroutine owns
|
||||
partitioning, database writes, and progress rendering; progress display must
|
||||
never block the workers.
|
||||
|
||||
`scan` writes nothing to stdout. The summary line on stderr reports the files
|
||||
seen this run broken down by disposition, plus skips:
|
||||
@@ -262,14 +534,17 @@ total.)
|
||||
|
||||
**`report` must never touch the filesystem being analyzed.** It does not stat,
|
||||
open, or otherwise access any path that appears in the records; its only I/O is
|
||||
reading the database and writing stdout/stderr. It must produce identical output
|
||||
reading the database, writing stdout/stderr, and the temporary file SQLite sorts
|
||||
in when the duplicate rows do not fit in memory. SQLite puts that file in
|
||||
`$SQLITE_TMPDIR` or `$TMPDIR` when set, otherwise in `/var/tmp` (or `/tmp`), and
|
||||
deletes it as soon as it has opened it. `report` must produce identical output
|
||||
whether or not the scanned filesystem is still mounted.
|
||||
|
||||
Processing:
|
||||
|
||||
- Records without hashes (size-unique when last scanned) are excluded: their
|
||||
- Records without a `content` hash (see "Database" above) are excluded: their
|
||||
content is unknown, so they are never reported as duplicates.
|
||||
- Group the remaining records by the key `(size, head_hash, tail_hash)`.
|
||||
- Group the remaining records by the key `(size, head, tail, content)`.
|
||||
- Every group with two or more paths is a duplicate group.
|
||||
- Within each group, sort paths lexicographically (byte order). The first path
|
||||
is the group's `first`; every other path is a `dupe`.
|
||||
@@ -288,6 +563,14 @@ first dupe size
|
||||
/srv/a/big.iso /srv/c/big-copy2.iso 4294967296
|
||||
```
|
||||
|
||||
Paths are raw bytes and may hold any byte except NUL, so the path columns
|
||||
(`first` and `dupe`) are escaped to keep every row one line of tab-separated
|
||||
fields: a backslash is written as `\\`, a tab as `\t`, a newline as `\n`, and a
|
||||
carriage return as `\r`. Every other byte is written unchanged, including bytes
|
||||
that are not valid UTF-8. Undoing those four escapes gives back the stored path.
|
||||
Grouping and ordering use the stored path, not the escaped one. The warnings
|
||||
`scan` prints on stderr are escaped the same way, so each warning is one line.
|
||||
|
||||
Summary to stderr: records read, number of duplicate groups, number of dupe
|
||||
files, and total reclaimable bytes (sum of `size` over all dupe rows) in human
|
||||
units.
|
||||
@@ -304,10 +587,10 @@ records, split on `/`.
|
||||
|
||||
Definitions:
|
||||
|
||||
- A file's **signature** is `(size, head_hash, tail_hash)` — mtime is
|
||||
informational and excluded. An unhashed record (empty hashes) has unknown
|
||||
- A file's **signature** is `(size, head, tail, content)` — mtime is
|
||||
informational and excluded. A record without a `content` hash has unknown
|
||||
content: its signature is treated as unique to that file, so a tree containing
|
||||
an unhashed file never compares equal to any other tree.
|
||||
such a file never compares equal to any other tree.
|
||||
- A directory's **digest** is a SHA-256 Merkle digest computed bottom-up:
|
||||
serialize the directory's child entries — for a file child, its name and
|
||||
signature; for a subdirectory child, its name and that subdirectory's digest —
|
||||
@@ -353,6 +636,9 @@ first dupe files size
|
||||
/srv/a/project /srv/backup/project 3417 104857600
|
||||
```
|
||||
|
||||
The `first` and `dupe` paths are escaped as described under "Report output
|
||||
format". The root directory's path is `/`.
|
||||
|
||||
Summary to stderr: records read, number of duplicate-tree groups, number of dupe
|
||||
trees, and total reclaimable bytes (sum of `size` over all dupe rows) in human
|
||||
units.
|
||||
@@ -365,9 +651,12 @@ Use the progress-bar library for all scan progress; rendering in the style of
|
||||
Each phase gets its own display, rendered the moment the phase starts — a scan
|
||||
must never look hung. Loading the existing-record index (`load`) and the walk
|
||||
have no known totals while running: show a live count, rate, and elapsed time
|
||||
(spinner-style, no percentage or ETA). The hash and update phases have exact
|
||||
totals — only files that actually need hashing appear in the hash total, so its
|
||||
ETA is meaningful. Required elements for the bars with known totals:
|
||||
(spinner-style, no percentage or ETA). The content phase's display (`content`)
|
||||
starts the same way, counting the records checked while SQLite finds the files
|
||||
to read and `lstat` checks them, then shows a bar once reading starts. The hash
|
||||
and update phases, and the content phase's reads, have exact totals — only files
|
||||
that actually need hashing appear in the hash and content totals, so their ETAs
|
||||
are meaningful. Required elements for the bars with known totals:
|
||||
|
||||
- elapsed time
|
||||
- estimated time remaining
|
||||
@@ -382,21 +671,57 @@ hash: [12345/98765] 12% |████ | 92 files/s elapsed 2:32 eta 17:54
|
||||
|
||||
Additional requirements:
|
||||
|
||||
- When stderr is not a TTY, do not emit ANSI redraws: print a plain one-line
|
||||
progress update no more often than every 5 seconds instead.
|
||||
- When stderr is not a terminal (a pipe, a file, `/dev/null`), do not emit ANSI
|
||||
redraws: print a plain one-line progress update the moment each phase starts,
|
||||
then no more often than every 5 seconds.
|
||||
- Progress updates are driven from the main goroutine and must be non-blocking
|
||||
with respect to the worker pool.
|
||||
with respect to the worker pool. On a terminal the spinner-style displays also
|
||||
redraw on their own several times a second, so their count and elapsed time
|
||||
stay current while a phase waits for its next item.
|
||||
- A warning printed during a phase always lands on a line of its own, never
|
||||
inside the progress display.
|
||||
- A bar whose phase stops short of its total, as an interrupted one does, is
|
||||
left as last drawn rather than filled up.
|
||||
- `report` and `trees` modes need no progress display, only their stderr
|
||||
summaries.
|
||||
|
||||
### Error handling and exit codes
|
||||
|
||||
- `0`: success, even if individual files were skipped with warnings.
|
||||
- `1`: fatal error (e.g., a `PATH` operand does not exist, the database cannot
|
||||
be created/opened/read/written, a missing database for `report`/`trees`,
|
||||
stdout write failure).
|
||||
- `2`: usage error (including `scan` with no `PATH` operand and `report`/`trees`
|
||||
with any positional argument).
|
||||
- `1`: fatal error (e.g., a `PATH` operand does not exist, another `scan` is
|
||||
already running against the same database, the database cannot be
|
||||
created/opened/read/written, a missing database for `report`/`trees`, stdout
|
||||
write failure), or a `scan` stopped by `SIGINT` or `SIGTERM` (see below).
|
||||
- `2`: usage error (including `scan` with no `PATH` operand, `scan` with
|
||||
`--workers` below 1, and `report`/`trees` with any positional argument).
|
||||
|
||||
A stdout write failure, such as a full disk, is reported in one line on stderr
|
||||
and exits 1. Two cases never reach sfdupes as a failed write:
|
||||
|
||||
- When the reader of a stdout pipe exits early, as in `sfdupes report | head`,
|
||||
the next write ends sfdupes with `SIGPIPE`, quietly and without a summary, the
|
||||
way `cat` or `sort` end. The shell reports the signal (status 141 in most
|
||||
shells), not exit 1.
|
||||
- When stdout is closed outright (`sfdupes report >&-`), the Go runtime opens
|
||||
`/dev/null` in its place before sfdupes starts, so the output is discarded and
|
||||
the run succeeds, as with `> /dev/null`.
|
||||
|
||||
`scan` stops cleanly on `SIGINT` (Ctrl-C) or `SIGTERM`. Its workers stop taking
|
||||
work, each finishing at most the directory listing or file it is reading; the
|
||||
progress display is finished; and the records it has hashed but not yet
|
||||
committed are committed, so the next scan does not hash them again. Apart from
|
||||
that commit it starts no further writes or deletions: records are deleted only
|
||||
after a complete walk, so those under paths an interrupted walk never reached
|
||||
are kept. The database is closed and the lock released as on any other exit, the
|
||||
line `scan: interrupted after N files` goes to stderr, N being the number of
|
||||
files the walk reached, and the exit code is 1. The next scan skips the records
|
||||
already written and converges as usual.
|
||||
|
||||
After the first signal `scan` stops catching them, so a second one ends it at
|
||||
once, as an uncaught signal does: the records not yet committed are lost, and
|
||||
the database is left valid, as when any scan dies (see "Database"). A `SIGINT`
|
||||
that `scan` inherits as ignored, as a script's background job does, stays
|
||||
ignored.
|
||||
|
||||
## Entrypoints
|
||||
|
||||
@@ -409,18 +734,14 @@ from any working directory, and may be invoked directly. The provided
|
||||
entrypoints are:
|
||||
|
||||
- `script/bootstrap` — install everything needed to build and develop this
|
||||
repository, idempotently, assuming nothing is present. `git`, `make`, `go`,
|
||||
and `node` come from the first of nix, apt, brew, or apk found on the host,
|
||||
and are presence-checked only; `node` is an unpinned host runtime like the
|
||||
rest, because nvm's prebuilt node is glibc-linked and does not run on this
|
||||
repo's musl/Alpine build image. The Markdown formatter itself — `prettier` —
|
||||
is pinned by `yarn.lock`'s integrity hash and installed with
|
||||
`yarn install --frozen-lockfile`. `golangci-lint` is deliberately **not**
|
||||
installed: it runs from a digest-pinned image via `script/lint` and never from
|
||||
a host install, so there is no host copy to drift from the pin. A missing
|
||||
repository, idempotently, assuming nothing is present. `git`, `make`, and `go`
|
||||
come from the first of nix, apt, brew, or apk found on the host, and are
|
||||
presence-checked only. `golangci-lint` and prettier are deliberately **not**
|
||||
installed: they run in Docker (see `script/lint` and `script/fmt`) and never
|
||||
from a host install, so there is no host copy to drift from the pin. A missing
|
||||
`docker` is warned about rather than installed or treated as fatal —
|
||||
everything except linting works without it. Ends with `go mod download` and
|
||||
the `yarn` install.
|
||||
everything except linting and formatting works without it. Ends with
|
||||
`go mod download`.
|
||||
- `script/setup` — make a fresh clone ready for development: runs
|
||||
`script/bootstrap`, then `script/install-precommit`.
|
||||
- `script/projectname` — print this project's name (`sfdupes`). Scripts that
|
||||
@@ -445,15 +766,22 @@ entrypoints are:
|
||||
entirely offline, until `go.mod` or `go.sum` changes and the download layer
|
||||
goes cold again. Because the daemon only ever sees a build context, this works
|
||||
when the docker daemon is remote and bind mounts are impossible.
|
||||
- `script/fmt` — format in place: `gofmt -s -w` for Go sources and `prettier`
|
||||
for Markdown (`--tab-width 4 --prose-wrap always`, the house settings, also
|
||||
carried in `.prettierrc`). prettier is the pinned devDependency in
|
||||
`package.json`/`yarn.lock`, installed by `script/bootstrap`.
|
||||
- `script/fmt-check` — the read-only counterpart of `script/fmt`: runs both
|
||||
checks, reports each independently so it is clear which failed, and exits
|
||||
non-zero if either found unformatted files instead of writing.
|
||||
- `script/fmt` — format in place: the Go sources with `gofmt -s -w`, and every
|
||||
Markdown file with prettier, at the settings in `.prettierrc` (4-space
|
||||
indents, prose wrapped at 80 columns). prettier is pinned by hash through
|
||||
`package.json` and `yarn.lock` and never installed on the host: this builds
|
||||
the `Dockerfile`'s `prettier` stage, a digest-pinned node image into which
|
||||
`yarn install --frozen-lockfile` installs it, tagged `sfdupes-prettier`, and
|
||||
runs that with the repository mounted, as the calling user. Needs `docker`,
|
||||
and because of the mount, unlike `script/lint`, a local docker daemon.
|
||||
- `script/fmt-check` — the read-only counterpart of `script/fmt`, with the
|
||||
repository mounted read-only: prints any unformatted file and exits non-zero
|
||||
instead of writing. gofmt and prettier both run every time, and each names
|
||||
itself when it fails. The `Dockerfile` runs the same two checks as gates: the
|
||||
gofmt check in its lint stage, prettier in its `markdown` stage.
|
||||
- `script/check` — run `script/test`, `script/lint`, and `script/fmt-check`, in
|
||||
that order. Modifies nothing. Needs `docker`, because `script/lint` does.
|
||||
that order. Modifies nothing. Needs `docker`, because `script/lint` and
|
||||
`script/fmt-check` do.
|
||||
- `script/docker` — build the Docker image, tagged with the name from
|
||||
`script/projectname`. The `Dockerfile` runs the gates as build steps, so this
|
||||
is also the check a developer or reviewer runs by hand.
|
||||
@@ -507,10 +835,11 @@ compile recipe:
|
||||
failure).
|
||||
- `make lint` — run `golangci-lint` with the repo config, in Docker (see
|
||||
`script/lint`); requires `docker`.
|
||||
- `make fmt` / `make fmt-check` — format Go and Markdown sources / verify both
|
||||
without writing.
|
||||
- `make fmt` / `make fmt-check` — format the Go sources and the Markdown /
|
||||
verify formatting without writing; requires `docker`, for prettier (see
|
||||
`script/fmt`).
|
||||
- `make check` — `test`, `lint`, and `fmt-check`; modifies nothing. Requires
|
||||
`docker`, via `lint`.
|
||||
`docker`, via `lint` and `fmt-check`.
|
||||
- `make docker` — build the Docker image, which runs the gates as build stages.
|
||||
- `make hooks` — install the pre-commit hook.
|
||||
- `make clean` — remove the binary.
|
||||
@@ -519,14 +848,14 @@ compile recipe:
|
||||
|
||||
All of the following, run in this directory, must pass:
|
||||
|
||||
1. `make check` passes (tests, lint, `gofmt`).
|
||||
1. `make check` passes (tests, lint, `gofmt`, prettier).
|
||||
2. `make docker` succeeds.
|
||||
3. Smoke test — create a throwaway tree in a temp dir (never test against real
|
||||
data):
|
||||
|
||||
```sh
|
||||
d=$(mktemp -d)
|
||||
export SFDUPES_DATABASE="$d/db.sqlite"
|
||||
export SFDUPES_DATABASE="$(mktemp -d)/db.sqlite"
|
||||
mkdir -p "$d/a" "$d/b"
|
||||
head -c 2000 /dev/urandom > "$d/a/one.bin"
|
||||
cp "$d/a/one.bin" "$d/b/copy.bin"
|
||||
@@ -554,8 +883,9 @@ All of the following, run in this directory, must pass:
|
||||
./sfdupes report
|
||||
```
|
||||
|
||||
(The scan database lives inside `$d` here purely for test hygiene; scanning
|
||||
`$d` therefore also records the SQLite file itself, which is harmless.)
|
||||
(The database lives in a temp directory of its own: inside `$d`, the scan
|
||||
would record it, and its empty lock file would join the `empty1`/`empty2`
|
||||
group.)
|
||||
|
||||
Expected from the first `report`: `one.bin`/`copy.bin`/`copy2.bin` form one
|
||||
group (two dupe rows, `first` is the lexicographically smallest path);
|
||||
@@ -584,8 +914,10 @@ Tracked in [TODO.md](TODO.md).
|
||||
|
||||
## Non-goals
|
||||
|
||||
- No full-content verification, no byte-for-byte compare, no deletion or linking
|
||||
of duplicates. The reports are advisory; acting on them is the user's job.
|
||||
- No byte-for-byte compare, and no deletion or linking of duplicates. Files that
|
||||
match are compared by a SHA-256 of the whole file below 50 MiB, and only by
|
||||
samples at 50 MiB and over. The reports are advisory; acting on them is the
|
||||
user's job.
|
||||
- No persistence beyond the SQLite database described above; no export/import
|
||||
formats.
|
||||
- No daemon or filesystem watcher; scheduling rescans is cron's job.
|
||||
|
||||
@@ -28,9 +28,102 @@
|
||||
|
||||
# Completed Steps
|
||||
|
||||
- restore Markdown formatting in `script/fmt`/`fmt-check` and reformat all
|
||||
Markdown to the house prettier settings (2026-09-21, closes
|
||||
- `make fmt` and `make fmt-check` run prettier over all Markdown, in Docker, and
|
||||
CI checks it; all Markdown reformatted (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/19)
|
||||
|
||||
- `scan` rejects `--workers` below 1 as a usage error instead of running
|
||||
single-threaded (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/10)
|
||||
|
||||
- a test fails when either walk cancellation check in `scan.go` is removed
|
||||
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/81)
|
||||
|
||||
- test that `scan` refuses a database with another schema version (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/64)
|
||||
|
||||
- correct four inaccurate comments in `cancel_test.go` and rename
|
||||
`walkCancelInFlightDirs` to `walkCancelInFlightFiles` (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/33)
|
||||
|
||||
- test the `-x` filesystem-boundary rules in `subdirJob` (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/17)
|
||||
|
||||
- `scan` creates the schema in one transaction; a version-0 database with a
|
||||
`files` table is refused with a clear schema-version error (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/11)
|
||||
|
||||
- README documents install, Docker, a daily cron scan and how to read and check
|
||||
the reports (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/54)
|
||||
|
||||
- the `Dockerfile` build stage keeps the Go module cache out of `builder`'s home
|
||||
and copies the sources with `--chown`, so no `chown -R` walks them
|
||||
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/43)
|
||||
|
||||
- `--version` prints `sfdupes VERSION` to stdout; README documents it and
|
||||
`--help` (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/15)
|
||||
|
||||
- `scan` stops cleanly on `SIGINT` or `SIGTERM`: commits what it has hashed,
|
||||
deletes nothing more, exits 1 (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/5)
|
||||
|
||||
- `report` and `trees` stream the records instead of holding them all in memory;
|
||||
the schema gains the `files_signature` index (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/14)
|
||||
|
||||
- progress prints at once on a non-terminal, uses a real terminal test, and
|
||||
prints warnings through a spinner instead of racing its redraw (2026-10-03,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/13)
|
||||
|
||||
- warn about and skip symlink, socket, FIFO, device and `.zfs` operands, keeping
|
||||
the records beneath them (2026-10-03,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/9)
|
||||
|
||||
- `scan` holds a lock on a lock file beside the database for its whole run, so a
|
||||
second `scan` fails at once with exit 1 (2026-10-03,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/53)
|
||||
|
||||
- test stdout write failures in `report` and `trees`; README states that
|
||||
`| head` ends sfdupes by `SIGPIPE` and `>&-` writes to `/dev/null`
|
||||
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/30)
|
||||
|
||||
- `report` and `trees` open the database read-only, and `scan` leaves it out of
|
||||
WAL mode, so reading needs only read access (2026-10-03, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/8)
|
||||
|
||||
- escape tabs, newlines, carriage returns and backslashes in report, trees and
|
||||
warning paths; the root directory's path is `/` (2026-10-03,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/7)
|
||||
|
||||
- stamp the git tag or short commit in a plain `docker build .` instead of `dev`
|
||||
(2026-10-02, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now sends
|
||||
`.git`, without `.git/config`, and the `Dockerfile` build stage takes the
|
||||
`VERSION` build argument when one is given, otherwise
|
||||
`git describe --tags --always` of that `.git`. The build fails if the context
|
||||
carries `.git` and the version still comes out empty, `dev` or `unknown`. The
|
||||
CI checkout step fetches the full history (`fetch-depth: 0`) so CI sees the
|
||||
tag and stamps the same value as `make build`.
|
||||
|
||||
- replace the 1 KiB end-window sampling with the head/tail plus content-hash
|
||||
ladder (2026-09-22, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is hashed in
|
||||
full and compared directly, with no end-window step — its `head`, `tail`, and
|
||||
`content` all hold the whole-file hash. A file at 10 MiB or above gets only
|
||||
the 64 KiB `head` and `tail` in the hash phase; a new content phase, after the
|
||||
update phase, reads it for its `content` hash — the whole file below 50 MiB,
|
||||
gigabyte-spaced 1 MiB samples at or above — only when its size, `head`, and
|
||||
`tail` match another record's, from the same scan or stored by an earlier one,
|
||||
so a stored file gains its content hash when it gains a match. A file that is
|
||||
gone or has changed since its record was written is not read. The `content`
|
||||
column is part of the version 1 schema. `report` and `trees` group by the
|
||||
extended signature and leave out any record without a `content` hash, so the
|
||||
ladder is applied across the whole database. README "Duplicate detection"
|
||||
documents every rung including the probabilistic large-file path.
|
||||
|
||||
- remove the dead `files.dat` references from `Makefile`, `.gitignore` and
|
||||
`.dockerignore` (2026-09-21, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/22)
|
||||
|
||||
- fix the lint-image pin comments and `FROM` form in `Dockerfile` and
|
||||
`Dockerfile.lint` (2026-08-10, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/25): dropped the false
|
||||
|
||||
+341
-37
@@ -4,20 +4,35 @@ import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/signal"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// poolUnwind bounds how long a goroutine is given to leave a pool
|
||||
// after its context is cancelled. Only a failing run ever waits this
|
||||
// long: a pool that ignored its cancellation parks forever, and this
|
||||
// is what turns that into a failed assertion instead of a suite that
|
||||
// hangs until the test binary's own timeout.
|
||||
// This file gathers the tests for scan cancellation and worker-pool
|
||||
// unwinding. Everything it exercises lives in scan.go, so by the repo's
|
||||
// convention of one test file per source file it would belong in
|
||||
// scan_test.go. It is kept separate on purpose: cancellation behaviour
|
||||
// cuts across both the walk pool and the hash pool as a single concern,
|
||||
// and scan_test.go is already over 1,600 lines. That is the deliberate
|
||||
// exception the convention otherwise expects to be stated.
|
||||
|
||||
// poolUnwind bounds how long a test waits for a cancellation to take
|
||||
// effect: for a goroutine to return or a channel to close once its
|
||||
// context is cancelled, or for a signal to cancel the scan's context.
|
||||
// Only a failing run waits this long, and the bound is what makes that
|
||||
// failure an assertion instead of a hang. A call made without it, as
|
||||
// most of this file's scans are, has no bound: a regression that parks
|
||||
// it is caught only as the test binary's own timeout.
|
||||
const poolUnwind = 2 * time.Second
|
||||
|
||||
// walkClock is a context whose cancellation is driven by the scan's
|
||||
@@ -28,9 +43,14 @@ const poolUnwind = 2 * time.Second
|
||||
//
|
||||
// The accounting behind the n chosen by each test: every blocking
|
||||
// channel operation in the walk selects on Done, so the walk spends
|
||||
// one consultation per file event plus a couple per directory, while
|
||||
// the index load that runs ahead of it spends a small fixed number
|
||||
// (three) whatever the record count.
|
||||
// one consultation per file event plus a couple per directory. The
|
||||
// index load that runs ahead of it also consults Done, but a bounded
|
||||
// number of times that does not grow with the record count. The tests
|
||||
// depend on that property, not on the bound's exact value: each test
|
||||
// sets n from the consultations of the walk, plus those of the hash
|
||||
// phase when it cancels mid-hash, far from both ends of the phase it
|
||||
// interrupts, so the cancellation lands inside that phase whatever the
|
||||
// record count.
|
||||
type walkClock struct {
|
||||
n int64
|
||||
seen atomic.Int64
|
||||
@@ -90,7 +110,10 @@ const (
|
||||
walkCancelFilesPerDir = 20
|
||||
walkCancelFiles = walkCancelDirs * walkCancelFilesPerDir
|
||||
walkCancelWorkers = 4
|
||||
walkCancelInFlightDirs = walkCancelWorkers * walkCancelFilesPerDir
|
||||
// The most files the walkCancelWorkers directories already in
|
||||
// flight when the scan is cancelled can still emit, at
|
||||
// walkCancelFilesPerDir each. A file count, not a directory count.
|
||||
walkCancelInFlightFiles = walkCancelWorkers * walkCancelFilesPerDir
|
||||
)
|
||||
|
||||
// walkCancelAtDone is the consultation on which the fixture's context
|
||||
@@ -150,9 +173,13 @@ func assertRecordsIntact(t *testing.T, db *sql.DB, before []string) {
|
||||
// Every one of those records would look vanished to the update phase.
|
||||
// The guard is what stops the scan there, and this test is what
|
||||
// notices if it stops doing so: deleting the guard, or making it
|
||||
// unreachable, makes the scan carry its truncated view into a later
|
||||
// phase and fail there instead, with a wrapped error rather than the
|
||||
// bare cancellation.
|
||||
// unreachable, makes the scan carry its truncated view into the update
|
||||
// phase, which counts every record the walk never reached for removal.
|
||||
//
|
||||
// The syncScan call here is not bounded by poolUnwind: a regression
|
||||
// that left a worker pool parked would hang it, and that regression is
|
||||
// caught only by the test binary's own timeout, not by a quick
|
||||
// assertion.
|
||||
//
|
||||
//nolint:paralleltest // counts goroutines: must not run beside others
|
||||
func TestSyncScanCancelledMidWalkKeepsRecords(t *testing.T) {
|
||||
@@ -182,10 +209,10 @@ func TestSyncScanCancelledMidWalkKeepsRecords(t *testing.T) {
|
||||
|
||||
// assertWalkGuardAborted checks that the scan stopped at the post-walk
|
||||
// guard: with a census that is neither empty (the walk really ran)
|
||||
// nor complete (it really was cut short), and with the guard's own
|
||||
// bare cancellation as the error. A wrapped error means the partial
|
||||
// census was carried past the guard into the hash or update phase,
|
||||
// which is the failure this test exists to catch.
|
||||
// nor complete (it really was cut short), and with no record counted
|
||||
// for removal. A removal count means the partial census was carried
|
||||
// past the guard into the update phase, which is the failure this test
|
||||
// exists to catch.
|
||||
func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
|
||||
t.Helper()
|
||||
|
||||
@@ -194,12 +221,6 @@ func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
|
||||
err, context.Canceled)
|
||||
}
|
||||
|
||||
if errors.Unwrap(err) != nil {
|
||||
t.Errorf("syncScan reported %q, want the guard's bare "+
|
||||
"cancellation: a wrapped error means the truncated census "+
|
||||
"reached a later phase", err)
|
||||
}
|
||||
|
||||
if st.unchanged == 0 {
|
||||
t.Fatalf("stats = %+v: the census is empty, so the walk never "+
|
||||
"ran and the guard was reached for the wrong reason", st)
|
||||
@@ -211,10 +232,11 @@ func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
|
||||
}
|
||||
|
||||
// The workers drop every directory still queued once the scan is
|
||||
// cancelled, so only the directories already in flight can add to
|
||||
// the census after the fact. A census beyond that bound would mean
|
||||
// the cancellation was not observed where it should have been.
|
||||
limit := walkCancelAtDone + walkCancelInFlightDirs
|
||||
// cancelled, so only the files in the directories already in flight
|
||||
// can add to the census after the fact. A census beyond that bound
|
||||
// would mean the cancellation was not observed where it should have
|
||||
// been.
|
||||
limit := walkCancelAtDone + walkCancelInFlightFiles
|
||||
if st.unchanged > limit {
|
||||
t.Errorf("census covers %d files, want at most %d: the walk kept "+
|
||||
"taking directories off the queue after cancellation",
|
||||
@@ -260,6 +282,242 @@ func TestSyncScanCancelledBeforeLoadIndex(t *testing.T) {
|
||||
assertRecordsIntact(t, db, before)
|
||||
}
|
||||
|
||||
// hashCancelAtDone is the consultation on which the mid-hash test's
|
||||
// context cancels itself. The walk of buildWalkCancelTree spends about
|
||||
// one per file and three per directory, and the hash phase then one per
|
||||
// file hashed, so this lands about half way through the hash phase.
|
||||
const hashCancelAtDone = walkCancelFiles + 3*walkCancelDirs +
|
||||
walkCancelFiles/2
|
||||
|
||||
// TestSyncScanCancelledMidHashKeepsHashedRecords cancels a first scan
|
||||
// part-way through its hash phase. The fixture holds fewer files than a
|
||||
// batch, so every file hashed is still waiting to be committed: the scan
|
||||
// must commit them all before it returns, and the next scan must hash
|
||||
// only the rest.
|
||||
func TestSyncScanCancelledMidHashKeepsHashedRecords(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := buildWalkCancelTree(t)
|
||||
db := openTestDB(t)
|
||||
|
||||
st, err := syncScan(newWalkClock(hashCancelAtDone), db,
|
||||
[]string{dir}, walkCancelWorkers, false)
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("syncScan cancelled mid-hash = %v, want %v",
|
||||
err, context.Canceled)
|
||||
}
|
||||
|
||||
if st.walked != walkCancelFiles || st.added == 0 ||
|
||||
st.added >= walkCancelFiles {
|
||||
t.Fatalf("stats = %+v: want the walk complete and the hash phase "+
|
||||
"cut short", st)
|
||||
}
|
||||
|
||||
if got := len(dbRecords(t, db)); got != st.added {
|
||||
t.Errorf("%d records after the cancelled scan, want the %d it hashed",
|
||||
got, st.added)
|
||||
}
|
||||
|
||||
hashed := st.added
|
||||
|
||||
st = syncTree(t, db, dir)
|
||||
if st.added != walkCancelFiles-hashed || st.unchanged != hashed {
|
||||
t.Errorf("next scan stats = %+v, want %d added %d unchanged",
|
||||
st, walkCancelFiles-hashed, hashed)
|
||||
}
|
||||
}
|
||||
|
||||
// storedPaths opens the database at path as report does, which fails
|
||||
// unless it is a valid database, and returns its records' paths.
|
||||
func storedPaths(t *testing.T, path string) []string {
|
||||
t.Helper()
|
||||
|
||||
db, err := openReportDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
defer func() { _ = db.Close() }()
|
||||
|
||||
return recordPaths(dbRecords(t, db))
|
||||
}
|
||||
|
||||
// TestRunScanInterrupted calls the scan entrypoint with a context that
|
||||
// is already cancelled, as when a signal arrives at once. It must return
|
||||
// errInterrupted promptly with its one line on stderr, leave the
|
||||
// database valid and as it was, and leave nothing in the way of the
|
||||
// next scan, which must bring the database up to date.
|
||||
func TestRunScanInterrupted(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
stderr := captureStderr(t)
|
||||
dir := buildSmokeTree(t)
|
||||
|
||||
err := runScan(t.Context(), []string{dir}, walkCancelWorkers, false)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
before := storedPaths(t, path)
|
||||
|
||||
// A vanished file and a new one: the interrupted scan records
|
||||
// neither.
|
||||
gone := filepath.Join(dir, "a", "unique.bin")
|
||||
|
||||
err = os.Remove(gone)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
added := writeFile(t, dir, "a/new.bin", pattern(50, 10))
|
||||
shown := len(stderr())
|
||||
done := make(chan struct{})
|
||||
|
||||
go func() {
|
||||
defer close(done)
|
||||
|
||||
err = runScan(cancelledContext(t), []string{dir}, walkCancelWorkers,
|
||||
false)
|
||||
}()
|
||||
|
||||
awaitReturn(t, done, "runScan")
|
||||
|
||||
if !errors.Is(err, errInterrupted) {
|
||||
t.Fatalf("runScan on a cancelled context = %v, want %v",
|
||||
err, errInterrupted)
|
||||
}
|
||||
|
||||
want := "scan: interrupted after 0 files\n"
|
||||
if got := stderr()[shown:]; got != want {
|
||||
t.Errorf("stderr = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
assertNoSidecars(t, path)
|
||||
|
||||
if got := storedPaths(t, path); !slices.Equal(got, before) {
|
||||
t.Errorf("records = %q after the interrupted scan, want %q",
|
||||
got, before)
|
||||
}
|
||||
|
||||
err = runScan(t.Context(), []string{dir}, walkCancelWorkers, false)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
got := storedPaths(t, path)
|
||||
if slices.Contains(got, gone) || !slices.Contains(got, added) {
|
||||
t.Errorf("records = %q after the next scan, want %q gone and %q "+
|
||||
"added", got, gone, added)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRunScanInterruptedMidHash interrupts the scan entrypoint part-way
|
||||
// through its hash phase, after the database is open. It must return
|
||||
// errInterrupted, release the lock, end stderr with its line counting
|
||||
// every file the walk reached, close the database out of WAL mode, and
|
||||
// keep the records it hashed.
|
||||
func TestRunScanInterruptedMidHash(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
stderr := captureStderr(t)
|
||||
dir := buildWalkCancelTree(t)
|
||||
|
||||
err := runScan(newWalkClock(hashCancelAtDone), []string{dir},
|
||||
walkCancelWorkers, false)
|
||||
if !errors.Is(err, errInterrupted) {
|
||||
t.Fatalf("runScan interrupted mid-hash = %v, want %v",
|
||||
err, errInterrupted)
|
||||
}
|
||||
|
||||
holdScanLock(t, path)
|
||||
|
||||
want := fmt.Sprintf("scan: interrupted after %d files\n", walkCancelFiles)
|
||||
if got := stderr(); !strings.HasSuffix(got, want) {
|
||||
t.Errorf("stderr = %q, want it to end with %q", got, want)
|
||||
}
|
||||
|
||||
assertNoSidecars(t, path)
|
||||
|
||||
db, err := openReportDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
defer func() { _ = db.Close() }()
|
||||
|
||||
// A plain close also removes the sidecars, but leaves WAL mode on.
|
||||
var mode string
|
||||
|
||||
err = db.QueryRowContext(t.Context(), "PRAGMA journal_mode").Scan(&mode)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if mode != "delete" {
|
||||
t.Errorf("journal mode = %q after the interrupted scan, want %q",
|
||||
mode, "delete")
|
||||
}
|
||||
|
||||
kept := len(dbRecords(t, db))
|
||||
if kept == 0 || kept >= walkCancelFiles {
|
||||
t.Errorf("%d records after the interrupted scan, want those it "+
|
||||
"hashed: some but not all of the %d files", kept, walkCancelFiles)
|
||||
}
|
||||
}
|
||||
|
||||
// TestInterruptContextCatchesSIGTERM sends SIGTERM to the test process
|
||||
// while the scan's handler is installed, and checks that it cancels the
|
||||
// scan's context.
|
||||
//
|
||||
//nolint:paralleltest // signals the whole process: must not run beside a scan
|
||||
func TestInterruptContextCatchesSIGTERM(t *testing.T) {
|
||||
// Caught here as well, so that a handler that misses SIGTERM fails
|
||||
// this test instead of ending the test process.
|
||||
caught := make(chan os.Signal, 1)
|
||||
signal.Notify(caught, syscall.SIGTERM)
|
||||
|
||||
defer signal.Stop(caught)
|
||||
|
||||
ctx, stop := interruptContext(t.Context())
|
||||
defer stop()
|
||||
|
||||
err := syscall.Kill(os.Getpid(), syscall.SIGTERM)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
case <-time.After(poolUnwind):
|
||||
t.Fatal("SIGTERM did not cancel the scan's context")
|
||||
}
|
||||
}
|
||||
|
||||
// TestCommitFullBatchKeepsFailedBatch checks that a full batch whose
|
||||
// commit fails, as it does once the scan is interrupted, stays in the
|
||||
// batch, so that syncScan's final commit saves it.
|
||||
func TestCommitFullBatchKeepsFailedBatch(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
s := &scanState{db: openTestDB(t)}
|
||||
for i := range updateBatchSize {
|
||||
s.batch = append(s.batch, scanRec{path: "/f" + strconv.Itoa(i)})
|
||||
}
|
||||
|
||||
err := s.commitFullBatch(cancelledContext(t))
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("commitFullBatch on a cancelled context = %v, want %v",
|
||||
err, context.Canceled)
|
||||
}
|
||||
|
||||
if len(s.batch) != updateBatchSize {
|
||||
t.Errorf("batch holds %d records after the failed commit, want %d",
|
||||
len(s.batch), updateBatchSize)
|
||||
}
|
||||
}
|
||||
|
||||
// drainClosed counts the values received from ch until it closes,
|
||||
// failing the test if it does not close within poolUnwind. A pool that
|
||||
// ignored its cancellation leaves its channel open with its goroutines
|
||||
@@ -328,20 +586,49 @@ func TestSendEventAbandonsBlockedSend(t *testing.T) {
|
||||
awaitReturn(t, done, "sendEvent")
|
||||
}
|
||||
|
||||
// TestWalkWorkersDropQueuedDirs checks that cancelled walk workers keep
|
||||
// reading jobs and drop the directories rather than stopping their
|
||||
// read: the range over jobs has to run out for the pool to tear down
|
||||
// and close its event stream.
|
||||
func TestWalkWorkersDropQueuedDirs(t *testing.T) {
|
||||
// TestWalkOneDirStopsWhenCancelled checks that a cancelled scan stops
|
||||
// reading a directory instead of going through the rest of its
|
||||
// entries. A walk that kept going would return the subdirectory below
|
||||
// to descend into. Unlike a file event, that return is not a send the
|
||||
// cancellation can abandon, so the test catches the regression every
|
||||
// time.
|
||||
func TestWalkOneDirStopsWhenCancelled(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
writeEmptyFiles(t, dir, walkCancelFilesPerDir)
|
||||
|
||||
err := os.Mkdir(filepath.Join(dir, "sub"), 0o750)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Unbuffered and unread: on a cancelled scan every send gives up.
|
||||
events := make(chan walkEvent)
|
||||
|
||||
subs := walkOneDir(cancelledContext(t), dirJob{path: dir}, false, events)
|
||||
if len(subs) != 0 {
|
||||
t.Errorf("cancelled walkOneDir returned %+v to descend into, "+
|
||||
"want none", subs)
|
||||
}
|
||||
}
|
||||
|
||||
// TestWalkWorkersDropQueuedDirs checks that cancelled walk workers keep
|
||||
// reading jobs and drop the directories rather than stopping their
|
||||
// read: the range over jobs has to run out for the pool to tear down
|
||||
// and close its event stream. The queued directory does not exist, so
|
||||
// a worker that walked it anyway would send a warning before
|
||||
// walkOneDir's own cancellation check could stop it. On a cancelled
|
||||
// scan that send delivers or gives up at random, so with 64 jobs
|
||||
// queued the regression has a one in 2^64 chance of passing.
|
||||
func TestWalkWorkersDropQueuedDirs(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
missing := filepath.Join(t.TempDir(), "missing")
|
||||
|
||||
jobs, _, events := startWalkWorkers(cancelledContext(t), 2, false)
|
||||
|
||||
for range 4 {
|
||||
jobs <- dirJob{path: dir}
|
||||
for range 64 {
|
||||
jobs <- dirJob{path: missing}
|
||||
}
|
||||
|
||||
close(jobs)
|
||||
@@ -412,7 +699,11 @@ func TestDispatchDirsClosesJobsWhenCancelled(t *testing.T) {
|
||||
|
||||
// TestFeedHashJobsClosesJobsWhenCancelled checks that the hash feeder
|
||||
// abandons the runs it has not queued yet and still closes the job
|
||||
// channel, which is what lets the workers' range terminate.
|
||||
// channel, which is what lets the workers' range terminate. The
|
||||
// receive on jobs below is not bounded: a feeder that returned without
|
||||
// closing jobs would leave that receive with no sender and no close, so
|
||||
// this regression is caught by the test binary's timeout rather than by
|
||||
// a bounded assertion.
|
||||
func TestFeedHashJobsClosesJobsWhenCancelled(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
@@ -440,6 +731,16 @@ func TestFeedHashJobsClosesJobsWhenCancelled(t *testing.T) {
|
||||
// nobody wants the hashes of — while still letting the range run out
|
||||
// so the pool tears down. The queued run names a file that does not
|
||||
// exist, so a worker that hashed it anyway would produce a result.
|
||||
//
|
||||
// hashWorker's other cancellation exit, abandoning the send of a
|
||||
// result, is reachable from the scan: stop cancels the pool before it
|
||||
// drains results, so a worker waiting on that send can leave through
|
||||
// it. The tests that stop a scan mid-hash, among them
|
||||
// TestScanHashWriteFailureUnwindsPool, reach it in some runs only,
|
||||
// depending on timing, and no test fails without it, since stop's
|
||||
// drain frees a waiting worker anyway. This test, for its part, catches
|
||||
// a removed drop check in some runs only: a worker that hashes the run
|
||||
// anyway then picks at random between sending the result and leaving.
|
||||
func TestHashWorkerDropsQueuedRuns(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
@@ -456,7 +757,7 @@ func TestHashWorkerDropsQueuedRuns(t *testing.T) {
|
||||
go func() {
|
||||
defer close(done)
|
||||
|
||||
hashWorker(cancelledContext(t), jobs, results)
|
||||
hashWorker(cancelledContext(t), jobs, results, hashSignature)
|
||||
}()
|
||||
|
||||
awaitReturn(t, done, "hashWorker")
|
||||
@@ -472,7 +773,10 @@ func TestHashWorkerDropsQueuedRuns(t *testing.T) {
|
||||
// TestHashPhaseCancelledReturnsContextError checks the result loop's
|
||||
// own exit: with the pool cancelled, no result will ever arrive, and
|
||||
// the loop must leave through the cancellation rather than wait for a
|
||||
// receive that cannot happen.
|
||||
// receive that cannot happen. This call is not bounded by poolUnwind: a
|
||||
// loop that dropped its cancellation case would block on that receive,
|
||||
// so the regression surfaces as the test binary's timeout rather than
|
||||
// as a bounded assertion.
|
||||
func TestHashPhaseCancelledReturnsContextError(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ import (
|
||||
"slices"
|
||||
"strconv"
|
||||
|
||||
"golang.org/x/sys/unix"
|
||||
// The pure-Go SQLite driver, registered as "sqlite"; keeps cgo
|
||||
// disabled.
|
||||
_ "modernc.org/sqlite"
|
||||
@@ -32,6 +33,11 @@ const schemaVersion = 1
|
||||
// scan.
|
||||
const dbDirPerm = 0o755
|
||||
|
||||
// lockFilePerm is the mode for the scan lock file. Anyone who can open
|
||||
// the file can hold the lock and keep every scan from running, so it
|
||||
// is open to its owner only.
|
||||
const lockFilePerm = 0o600
|
||||
|
||||
// createTableSQL is the schema applied to a fresh database. Paths are
|
||||
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8.
|
||||
const createTableSQL = `
|
||||
@@ -40,18 +46,26 @@ CREATE TABLE files (
|
||||
size INTEGER NOT NULL,
|
||||
mtime INTEGER NOT NULL,
|
||||
head TEXT NOT NULL,
|
||||
tail TEXT NOT NULL
|
||||
tail TEXT NOT NULL,
|
||||
content TEXT NOT NULL
|
||||
) WITHOUT ROWID
|
||||
`
|
||||
|
||||
// createIndexSQL indexes the records by signature, so report can have
|
||||
// SQLite group them without sorting the whole table.
|
||||
const createIndexSQL = `
|
||||
CREATE INDEX files_signature ON files (size, head, tail, content)
|
||||
`
|
||||
|
||||
// upsertSQL inserts one file record, replacing any existing record for
|
||||
// the same path.
|
||||
const upsertSQL = `
|
||||
INSERT INTO files (path, size, mtime, head, tail)
|
||||
VALUES (?, ?, ?, ?, ?)
|
||||
INSERT INTO files (path, size, mtime, head, tail, content)
|
||||
VALUES (?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT (path) DO UPDATE SET
|
||||
size = excluded.size, mtime = excluded.mtime,
|
||||
head = excluded.head, tail = excluded.tail
|
||||
head = excluded.head, tail = excluded.tail,
|
||||
content = excluded.content
|
||||
`
|
||||
|
||||
// errNoDatabase reports a missing database file for report/trees.
|
||||
@@ -62,6 +76,10 @@ var errNoDatabase = errors.New(
|
||||
// does not understand.
|
||||
var errSchemaVersion = errors.New("unsupported database schema version")
|
||||
|
||||
// errScanRunning reports that another scan holds the lock on the
|
||||
// database.
|
||||
var errScanRunning = errors.New("another scan is running")
|
||||
|
||||
// databasePath resolves the database location: SFDUPES_DATABASE when
|
||||
// set and non-empty, the compiled-in default otherwise.
|
||||
func databasePath() string {
|
||||
@@ -72,16 +90,24 @@ func databasePath() string {
|
||||
return defaultDatabasePath
|
||||
}
|
||||
|
||||
// openDB opens the SQLite database at path with WAL journaling and a
|
||||
// busy timeout, so a report can run while a cron scan is in progress.
|
||||
// It does not create or verify the schema.
|
||||
func openDB(path string) (*sql.DB, error) {
|
||||
dsn := "file:" + path +
|
||||
"?_pragma=busy_timeout(10000)" +
|
||||
// scanParams are the connection parameters for scan: read-write, with
|
||||
// WAL journaling and a busy timeout, so a report can run while a cron
|
||||
// scan is in progress. closeScanDatabase leaves WAL mode again.
|
||||
const scanParams = "_pragma=busy_timeout(10000)" +
|
||||
"&_pragma=journal_mode(WAL)" +
|
||||
"&_pragma=synchronous(NORMAL)"
|
||||
|
||||
db, err := sql.Open("sqlite", dsn)
|
||||
// reportParams are the connection parameters for report and trees:
|
||||
// read-only, with the same busy timeout. They set no journal mode,
|
||||
// because setting one is a write.
|
||||
const reportParams = "mode=ro" +
|
||||
"&_pragma=busy_timeout(10000)" +
|
||||
"&_pragma=query_only(1)"
|
||||
|
||||
// openDB opens the SQLite database at path with the connection
|
||||
// parameters params. It does not create or verify the schema.
|
||||
func openDB(path, params string) (*sql.DB, error) {
|
||||
db, err := sql.Open("sqlite", "file:"+path+"?"+params)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open database %s: %w", path, err)
|
||||
}
|
||||
@@ -94,6 +120,43 @@ func openDB(path string) (*sql.DB, error) {
|
||||
return db, nil
|
||||
}
|
||||
|
||||
// lockScanDatabase takes the lock that keeps a second scan off the
|
||||
// database at path: an exclusive flock(2) on the file beside it named
|
||||
// path with ".lock" appended, created along with the database's parent
|
||||
// directory if missing. A lock held by another scan fails at once
|
||||
// instead of waiting. The lock lasts until the returned file is closed
|
||||
// or the process ends. The file is never deleted: a scan that deleted
|
||||
// it would let the next scan lock a new file while another still holds
|
||||
// the old one.
|
||||
func lockScanDatabase(path string) (*os.File, error) {
|
||||
err := os.MkdirAll(filepath.Dir(path), dbDirPerm)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("create database directory: %w", err)
|
||||
}
|
||||
|
||||
lockPath := path + ".lock"
|
||||
|
||||
//nolint:gosec // the operator chooses the database path
|
||||
f, err := os.OpenFile(lockPath, os.O_RDWR|os.O_CREATE, lockFilePerm)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
err = unix.Flock(int(f.Fd()), unix.LOCK_EX|unix.LOCK_NB)
|
||||
if err != nil {
|
||||
_ = f.Close()
|
||||
|
||||
if errors.Is(err, unix.EWOULDBLOCK) {
|
||||
return nil, fmt.Errorf("%w (lock held on %s)",
|
||||
errScanRunning, lockPath)
|
||||
}
|
||||
|
||||
return nil, fmt.Errorf("lock %s: %w", lockPath, err)
|
||||
}
|
||||
|
||||
return f, nil
|
||||
}
|
||||
|
||||
// openScanDatabase opens the database for the scan subcommand, creating
|
||||
// the file, its parent directory, and the schema as needed.
|
||||
func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
|
||||
@@ -102,7 +165,7 @@ func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
|
||||
return nil, fmt.Errorf("create database directory: %w", err)
|
||||
}
|
||||
|
||||
db, err := openDB(path)
|
||||
db, err := openDB(path, scanParams)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -117,6 +180,24 @@ func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
|
||||
return db, nil
|
||||
}
|
||||
|
||||
// closeScanDatabase switches the database at path from WAL back to
|
||||
// rollback-journal mode and closes it. Out of WAL mode the database
|
||||
// file alone holds the whole database, so a reader needs no -wal or
|
||||
// -shm file beside it, nor write access to create them. The switch
|
||||
// fails while a report has the database open; the database then stays
|
||||
// in WAL mode, still readable, until a later scan closes it.
|
||||
func closeScanDatabase(ctx context.Context, db *sql.DB, path string) {
|
||||
// Runs on the way out of a cancelled scan too.
|
||||
_, err := db.ExecContext(context.WithoutCancel(ctx),
|
||||
"PRAGMA journal_mode = DELETE")
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "scan: database %s left in WAL mode: %v\n",
|
||||
path, err)
|
||||
}
|
||||
|
||||
_ = db.Close()
|
||||
}
|
||||
|
||||
// openReportDatabase opens an existing database for the report and
|
||||
// trees subcommands. A missing database file is an error directing the
|
||||
// user to run scan first; the schema version must match exactly.
|
||||
@@ -132,12 +213,18 @@ func openReportDatabase(ctx context.Context,
|
||||
return nil, fmt.Errorf("database: %w", err)
|
||||
}
|
||||
|
||||
db, err := openDB(path)
|
||||
db, err := openDB(path, reportParams)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
v, err := userVersion(ctx, db)
|
||||
if err == nil && v == 0 {
|
||||
// An empty database passes this check and fails the version
|
||||
// check below.
|
||||
err = checkUnversioned(ctx, db)
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
_ = db.Close()
|
||||
|
||||
@@ -164,6 +251,11 @@ func initSchema(ctx context.Context, db *sql.DB) error {
|
||||
|
||||
switch v {
|
||||
case 0:
|
||||
err = checkUnversioned(ctx, db)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
return createSchema(ctx, db)
|
||||
case schemaVersion:
|
||||
return nil
|
||||
@@ -173,20 +265,65 @@ func initSchema(ctx context.Context, db *sql.DB) error {
|
||||
}
|
||||
}
|
||||
|
||||
// checkUnversioned checks a database at user_version 0 before it is
|
||||
// taken for an empty one. createSchema creates the files table and
|
||||
// sets the version together, so a files table at version 0 was made by
|
||||
// something else. Adopting it could corrupt unrelated data, so that is
|
||||
// a schema-version error telling the operator to remove the file and
|
||||
// rescan.
|
||||
func checkUnversioned(ctx context.Context, db *sql.DB) error {
|
||||
var name string
|
||||
|
||||
err := db.QueryRowContext(ctx,
|
||||
"SELECT name FROM sqlite_master "+
|
||||
"WHERE type = 'table' AND name = 'files'").Scan(&name)
|
||||
|
||||
switch {
|
||||
case err == nil:
|
||||
return fmt.Errorf(
|
||||
"has a files table but no schema version; "+
|
||||
"remove the file and rescan: %w", errSchemaVersion)
|
||||
case errors.Is(err, sql.ErrNoRows):
|
||||
return nil
|
||||
default:
|
||||
return fmt.Errorf("check for files table: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
// createSchema applies the schema to a fresh database and stamps the
|
||||
// schema version.
|
||||
// schema version in one transaction, so a creation stopped partway, by
|
||||
// an interrupt or an error, leaves an empty database the next scan
|
||||
// sets up, never a files table at version 0, which checkUnversioned
|
||||
// refuses.
|
||||
func createSchema(ctx context.Context, db *sql.DB) error {
|
||||
_, err := db.ExecContext(ctx, createTableSQL)
|
||||
tx, err := db.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create schema: %w", err)
|
||||
}
|
||||
|
||||
_, err = db.ExecContext(ctx,
|
||||
defer func() { _ = tx.Rollback() }()
|
||||
|
||||
_, err = tx.ExecContext(ctx, createTableSQL)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create schema: %w", err)
|
||||
}
|
||||
|
||||
_, err = tx.ExecContext(ctx, createIndexSQL)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create schema: %w", err)
|
||||
}
|
||||
|
||||
_, err = tx.ExecContext(ctx,
|
||||
"PRAGMA user_version = "+strconv.Itoa(schemaVersion))
|
||||
if err != nil {
|
||||
return fmt.Errorf("set schema version: %w", err)
|
||||
}
|
||||
|
||||
err = tx.Commit()
|
||||
if err != nil {
|
||||
return fmt.Errorf("create schema: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -202,39 +339,113 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) {
|
||||
return v, nil
|
||||
}
|
||||
|
||||
// loadFileRows reads every record from the files table.
|
||||
func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
||||
// loadFileRows streams every record to fn in path order: byte order,
|
||||
// which is the order of the primary key, so SQLite does not sort.
|
||||
func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
|
||||
rows, err := db.QueryContext(ctx,
|
||||
"SELECT path, size, mtime, head, tail FROM files")
|
||||
"SELECT path, size, mtime, head, tail, content FROM files "+
|
||||
"ORDER BY path")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read records: %w", err)
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
var recs []scanRec
|
||||
|
||||
for rows.Next() {
|
||||
var (
|
||||
path []byte
|
||||
r scanRec
|
||||
)
|
||||
|
||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail)
|
||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
|
||||
&r.content)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read record: %w", err)
|
||||
return fmt.Errorf("read record: %w", err)
|
||||
}
|
||||
|
||||
r.path = string(path)
|
||||
recs = append(recs, r)
|
||||
fn(r)
|
||||
}
|
||||
|
||||
err = rows.Err()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read records: %w", err)
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
return recs, nil
|
||||
return nil
|
||||
}
|
||||
|
||||
// dupeRowsSQL selects every record in a duplicate group, with the
|
||||
// group's first path. A group is the records with a content hash that
|
||||
// share a size, head, tail, and content, when there are two or more of
|
||||
// them. The rows come in report order: groups by size descending, then
|
||||
// by first path, and each group's paths ascending.
|
||||
const dupeRowsSQL = `
|
||||
SELECT g.first, f.path, f.size
|
||||
FROM files AS f
|
||||
JOIN (
|
||||
SELECT size, head, tail, content, MIN(path) AS first
|
||||
FROM files
|
||||
WHERE content <> ''
|
||||
GROUP BY size, head, tail, content
|
||||
HAVING COUNT(*) > 1
|
||||
) AS g USING (size, head, tail, content)
|
||||
ORDER BY f.size DESC, g.first, f.path
|
||||
`
|
||||
|
||||
// loadDupeRows streams the rows of dupeRowsSQL to fn and returns the
|
||||
// number of records in the database. The count and the rows are read
|
||||
// in one transaction, so they agree while a scan is committing. An
|
||||
// error from fn stops the reading and is returned as it is.
|
||||
func loadDupeRows(ctx context.Context, db *sql.DB,
|
||||
fn func(first, path string, size int64) error,
|
||||
) (int, error) {
|
||||
// Everything goes through tx: the report connection is the only
|
||||
// one, so a query on db would wait for tx forever.
|
||||
tx, err := db.BeginTx(ctx, &sql.TxOptions{ReadOnly: true})
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = tx.Rollback() }()
|
||||
|
||||
var records int
|
||||
|
||||
err = tx.QueryRowContext(ctx, "SELECT COUNT(*) FROM files").Scan(&records)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
rows, err := tx.QueryContext(ctx, dupeRowsSQL)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
for rows.Next() {
|
||||
var (
|
||||
first, path []byte
|
||||
size int64
|
||||
)
|
||||
|
||||
err = rows.Scan(&first, &path, &size)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("read record: %w", err)
|
||||
}
|
||||
|
||||
err = fn(string(first), string(path), size)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
}
|
||||
|
||||
err = rows.Err()
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
return records, nil
|
||||
}
|
||||
|
||||
// loadFileMeta streams every record's path, size, mtime, and whether
|
||||
@@ -275,6 +486,62 @@ func loadFileMeta(ctx context.Context, db *sql.DB,
|
||||
return nil
|
||||
}
|
||||
|
||||
// contentCandidatesSQL selects every record of at least headTailMin
|
||||
// bytes whose size, head, and tail equal another record's, in each
|
||||
// group (the records sharing a size, head, and tail) where at least one
|
||||
// record has no content hash, with whether each record has one. SQLite
|
||||
// does the grouping, so no other record's hashes are loaded into
|
||||
// memory; the rows come ordered by size, head, and tail, so each
|
||||
// group's rows arrive together.
|
||||
const contentCandidatesSQL = `
|
||||
SELECT f.path, f.size, f.mtime, f.head, f.tail, f.content <> ''
|
||||
FROM files AS f
|
||||
JOIN (
|
||||
SELECT size, head, tail
|
||||
FROM files
|
||||
WHERE size >= ? AND head <> ''
|
||||
GROUP BY size, head, tail
|
||||
HAVING COUNT(*) > 1 AND SUM(content = '') > 0
|
||||
) AS g USING (size, head, tail)
|
||||
ORDER BY size, head, tail
|
||||
`
|
||||
|
||||
// loadContentCandidates streams the rows of contentCandidatesSQL to fn:
|
||||
// each record, without its content hash, and whether it has one.
|
||||
func loadContentCandidates(ctx context.Context, db *sql.DB,
|
||||
fn func(r scanRec, hashed bool),
|
||||
) error {
|
||||
rows, err := db.QueryContext(ctx, contentCandidatesSQL, headTailMin)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
for rows.Next() {
|
||||
var (
|
||||
path []byte
|
||||
r scanRec
|
||||
hashed int64
|
||||
)
|
||||
|
||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, &hashed)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read record: %w", err)
|
||||
}
|
||||
|
||||
r.path = string(path)
|
||||
fn(r, hashed != 0)
|
||||
}
|
||||
|
||||
err = rows.Err()
|
||||
if err != nil {
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// updateBatchSize is the number of record changes committed per
|
||||
// transaction during the update pass. The filesystem is authoritative
|
||||
// and the database an eventually-consistent reflection of it, so
|
||||
@@ -348,7 +615,7 @@ func execUpserts(ctx context.Context, tx *sql.Tx, upserts []scanRec,
|
||||
|
||||
for _, r := range upserts {
|
||||
_, err = st.ExecContext(ctx,
|
||||
[]byte(r.path), r.size, r.mtime, r.head, r.tail)
|
||||
[]byte(r.path), r.size, r.mtime, r.head, r.tail, r.content)
|
||||
if err != nil {
|
||||
return fmt.Errorf("upsert %s: %w", r.path, err)
|
||||
}
|
||||
|
||||
+142
-29
@@ -5,6 +5,7 @@ import (
|
||||
"database/sql"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strings"
|
||||
@@ -72,9 +73,78 @@ func TestOpenScanDatabaseCreates(t *testing.T) {
|
||||
|
||||
defer func() { _ = db.Close() }()
|
||||
|
||||
recs, err := loadFileRows(t.Context(), db)
|
||||
if err != nil || len(recs) != 0 {
|
||||
t.Fatalf("loadFileRows = %v, %v; want empty, nil", recs, err)
|
||||
if recs := dbRecords(t, db); len(recs) != 0 {
|
||||
t.Fatalf("records = %v, want none", recs)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOpenDatabaseUnversionedForeign(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// A database that has a files table but user_version 0, written by
|
||||
// some other tool. report, trees and scan must refuse it with the
|
||||
// schema-version error, not adopt it and not emit a raw SQLite
|
||||
// "table files already exists".
|
||||
path := testDBPath(t)
|
||||
|
||||
db, err := sql.Open("sqlite", path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
_, err = db.ExecContext(t.Context(), "CREATE TABLE files (x INTEGER)")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
_ = db.Close()
|
||||
|
||||
_, err = openReportDatabase(t.Context(), path)
|
||||
if !errors.Is(err, errSchemaVersion) ||
|
||||
!strings.Contains(err.Error(), "remove the file and rescan") {
|
||||
t.Fatalf("report: err = %v, want errSchemaVersion telling the "+
|
||||
"operator to remove the file and rescan", err)
|
||||
}
|
||||
|
||||
_, err = openScanDatabase(t.Context(), path)
|
||||
if !errors.Is(err, errSchemaVersion) ||
|
||||
!strings.Contains(err.Error(), "remove the file and rescan") {
|
||||
t.Fatalf("scan: err = %v, want errSchemaVersion telling the "+
|
||||
"operator to remove the file and rescan", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSchemaCreationStoppedPartway(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// A first scan stopped while creating the schema must leave a
|
||||
// database the next scan accepts. max_page_count(2) leaves room for
|
||||
// the files table but not its index, so schema creation fails right
|
||||
// after CREATE TABLE, a point an interrupt could also stop it at.
|
||||
path := testDBPath(t)
|
||||
|
||||
db, err := openDB(path, scanParams+"&_pragma=max_page_count(2)")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = initSchema(t.Context(), db)
|
||||
_ = db.Close()
|
||||
|
||||
if err == nil {
|
||||
t.Fatal("initSchema with no room for the index succeeded")
|
||||
}
|
||||
|
||||
db, err = openScanDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatalf("next scan: %v", err)
|
||||
}
|
||||
|
||||
defer func() { _ = db.Close() }()
|
||||
|
||||
v, err := userVersion(t.Context(), db)
|
||||
if err != nil || v != schemaVersion {
|
||||
t.Fatalf("userVersion = %d, %v; want %d, nil", v, err, schemaVersion)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -87,9 +157,11 @@ func TestOpenReportDatabaseMissing(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestOpenReportDatabaseVersionMismatch(t *testing.T) {
|
||||
func TestOpenDatabaseVersionMismatch(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// A database stamped with a schema version other than 0 and
|
||||
// schemaVersion. report, trees and scan must all refuse it.
|
||||
path := testDBPath(t)
|
||||
|
||||
db, err := openScanDatabase(t.Context(), path)
|
||||
@@ -106,7 +178,12 @@ func TestOpenReportDatabaseVersionMismatch(t *testing.T) {
|
||||
|
||||
_, err = openReportDatabase(t.Context(), path)
|
||||
if !errors.Is(err, errSchemaVersion) {
|
||||
t.Fatalf("err = %v, want errSchemaVersion", err)
|
||||
t.Fatalf("report: err = %v, want errSchemaVersion", err)
|
||||
}
|
||||
|
||||
_, err = openScanDatabase(t.Context(), path)
|
||||
if !errors.Is(err, errSchemaVersion) {
|
||||
t.Fatalf("scan: err = %v, want errSchemaVersion", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -130,16 +207,63 @@ func TestOpenReportDatabaseOK(t *testing.T) {
|
||||
_ = db.Close()
|
||||
}
|
||||
|
||||
func TestCloseScanDatabaseWhileReportOpen(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// A report holding the database open stops scan from taking it out
|
||||
// of WAL mode. The -wal and -shm files must then stay beside it, so
|
||||
// that a later report still needs only read access.
|
||||
path := testDBPath(t)
|
||||
|
||||
scanDB, err := openScanDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
reportDB, err := openReportDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
closeScanDatabase(t.Context(), scanDB, path)
|
||||
|
||||
_ = reportDB.Close()
|
||||
|
||||
_, err = os.Stat(path + "-wal")
|
||||
if err != nil {
|
||||
t.Fatalf("no -wal left: the switch out of WAL mode was not "+
|
||||
"stopped: %v", err)
|
||||
}
|
||||
|
||||
makeReadOnly(t, path)
|
||||
|
||||
reportDB, err = openReportDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatalf("openReportDatabase: %v", err)
|
||||
}
|
||||
|
||||
defer func() { _ = reportDB.Close() }()
|
||||
|
||||
err = loadFileRows(t.Context(), reportDB, func(scanRec) {})
|
||||
if err != nil {
|
||||
t.Fatalf("loadFileRows: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyChangesRoundTrip(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := openTestDB(t)
|
||||
|
||||
// Paths may contain tabs and newlines; the database must store
|
||||
// them byte-exactly.
|
||||
// them byte-exactly. Every hash, content included, comes back as
|
||||
// written.
|
||||
recs := []scanRec{
|
||||
{size: 2, mtime: 20, head: "h2", tail: "t2", path: "/a/tab\tnew\nline"},
|
||||
{size: 1, mtime: 10, head: "h1", tail: "t1", path: "/a/x"},
|
||||
{
|
||||
size: 2, mtime: 20, head: "h2", tail: "t2", content: "c2",
|
||||
path: "/a/tab\tnew\nline",
|
||||
},
|
||||
{size: 1, mtime: 10, head: "h1", tail: "t1", content: "c1", path: "/a/x"},
|
||||
}
|
||||
|
||||
err := applyChanges(t.Context(), db, recs, nil,
|
||||
@@ -148,22 +272,17 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
||||
t.Fatalf("applyChanges: %v", err)
|
||||
}
|
||||
|
||||
got, err := loadFileRows(t.Context(), db)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
slices.SortFunc(got, func(a, b scanRec) int {
|
||||
return strings.Compare(a.path, b.path)
|
||||
})
|
||||
|
||||
// The records come back in path order, which is the order of recs.
|
||||
got := dbRecords(t, db)
|
||||
if !slices.Equal(got, recs) {
|
||||
t.Fatalf("rows = %+v, want %+v", got, recs)
|
||||
}
|
||||
|
||||
// An upsert for an existing path updates in place; a delete
|
||||
// removes exactly its path.
|
||||
upd := scanRec{size: 3, mtime: 30, head: "h3", tail: "t3", path: "/a/x"}
|
||||
upd := scanRec{
|
||||
size: 3, mtime: 30, head: "h3", tail: "t3", content: "c3", path: "/a/x",
|
||||
}
|
||||
|
||||
err = applyChanges(t.Context(), db, []scanRec{upd},
|
||||
[]string{"/a/tab\tnew\nline"}, newProgress("update", 2))
|
||||
@@ -171,11 +290,7 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
||||
t.Fatalf("applyChanges: %v", err)
|
||||
}
|
||||
|
||||
got, err = loadFileRows(t.Context(), db)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
got = dbRecords(t, db)
|
||||
if len(got) != 1 || got[0] != upd {
|
||||
t.Fatalf("rows = %+v, want just %+v", got, upd)
|
||||
}
|
||||
@@ -204,9 +319,8 @@ func TestApplyChangesBatching(t *testing.T) {
|
||||
t.Fatalf("applyChanges: %v", err)
|
||||
}
|
||||
|
||||
got, err := loadFileRows(t.Context(), db)
|
||||
if err != nil || len(got) != n {
|
||||
t.Fatalf("loadFileRows = %d rows, %v; want %d", len(got), err, n)
|
||||
if got := dbRecords(t, db); len(got) != n {
|
||||
t.Fatalf("records = %d, want %d", len(got), n)
|
||||
}
|
||||
|
||||
deletes := make([]string, 0, n)
|
||||
@@ -220,8 +334,7 @@ func TestApplyChangesBatching(t *testing.T) {
|
||||
t.Fatalf("applyChanges deletes: %v", err)
|
||||
}
|
||||
|
||||
got, err = loadFileRows(t.Context(), db)
|
||||
if err != nil || len(got) != 0 {
|
||||
t.Fatalf("loadFileRows = %d rows, %v; want 0", len(got), err)
|
||||
if got := dbRecords(t, db); len(got) != 0 {
|
||||
t.Fatalf("records = %d, want 0", len(got))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,6 +5,8 @@ go 1.25.7
|
||||
require (
|
||||
github.com/schollz/progressbar/v3 v3.19.1
|
||||
github.com/spf13/cobra v1.10.2
|
||||
golang.org/x/sys v0.46.0
|
||||
golang.org/x/term v0.44.0
|
||||
modernc.org/sqlite v1.54.0
|
||||
)
|
||||
|
||||
@@ -18,8 +20,6 @@ require (
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
||||
github.com/rivo/uniseg v0.4.7 // indirect
|
||||
github.com/spf13/pflag v1.0.9 // indirect
|
||||
golang.org/x/sys v0.46.0 // indirect
|
||||
golang.org/x/term v0.44.0 // indirect
|
||||
modernc.org/libc v1.74.1 // indirect
|
||||
modernc.org/mathutil v1.7.1 // indirect
|
||||
modernc.org/memory v1.11.0 // indirect
|
||||
|
||||
@@ -1,16 +1,21 @@
|
||||
// Command sfdupes quickly identifies candidate duplicate files across
|
||||
// very large filesystems without reading full file contents. Files are
|
||||
// considered duplicates when they have identical size, identical SHA-256
|
||||
// of their first 1024 bytes, and identical SHA-256 of their last 1024
|
||||
// bytes. scan maintains a persistent SQLite database of file signatures
|
||||
// (SFDUPES_DATABASE, default /var/lib/sfdupes/db.sqlite) that the
|
||||
// reporting subcommands read.
|
||||
// very large filesystems without reading every byte of every file.
|
||||
// Files are considered duplicates when their sizes are equal and they
|
||||
// agree on a short ladder of SHA-256 hashes. A file under 10 MiB is
|
||||
// hashed in full. A larger file is compared on the hashes of its first
|
||||
// and last 64 KiB, and only when those match another file's is its
|
||||
// content hash computed and compared: of the whole file when it is
|
||||
// under 50 MiB, or of gigabyte-spaced 1 MiB samples when it is 50 MiB
|
||||
// or larger. scan maintains a persistent SQLite database of file
|
||||
// signatures (SFDUPES_DATABASE, default /var/lib/sfdupes/db.sqlite)
|
||||
// that the reporting subcommands read.
|
||||
//
|
||||
// Usage:
|
||||
//
|
||||
// sfdupes scan [--workers N] [-x] PATH...
|
||||
// sfdupes report > dupes.tsv
|
||||
// sfdupes trees > dupetrees.tsv
|
||||
// sfdupes --version
|
||||
//
|
||||
// See README.md for the complete specification.
|
||||
package main
|
||||
@@ -46,6 +51,10 @@ const (
|
||||
// cobra prints for it is the whole message.
|
||||
var errNoSubcommand = errors.New("no subcommand")
|
||||
|
||||
// errWorkersBelowOne is the usage error for a scan --workers value
|
||||
// below 1.
|
||||
var errWorkersBelowOne = errors.New("--workers must be at least 1")
|
||||
|
||||
// Version is the build version, injected at link time via -ldflags
|
||||
// (see the Makefile); "dev" for a plain go build.
|
||||
//
|
||||
@@ -53,22 +62,27 @@ var errNoSubcommand = errors.New("no subcommand")
|
||||
var Version = "dev"
|
||||
|
||||
func main() {
|
||||
os.Exit(run(os.Args[1:], os.Stderr))
|
||||
// Once the reader of a stdout pipe has gone, as in "sfdupes report |
|
||||
// head", the Go runtime ends the process with SIGPIPE on the next
|
||||
// write instead of returning an error (README "Error handling").
|
||||
// Registering for SIGPIPE with os/signal would change that.
|
||||
os.Exit(run(os.Args[1:], os.Stdout, os.Stderr))
|
||||
}
|
||||
|
||||
// run executes args against the command tree and returns the process
|
||||
// exit code. It is the program's single exit point: the subcommands
|
||||
// return their errors instead of exiting, so every deferred cleanup —
|
||||
// above all closing the database, which checkpoints the SQLite WAL —
|
||||
// runs before the process ends.
|
||||
func run(args []string, stderr io.Writer) int {
|
||||
// runs before the process ends. The report and trees subcommands write
|
||||
// their data to stdout.
|
||||
func run(args []string, stdout, stderr io.Writer) int {
|
||||
// A nil slice makes cobra fall back to os.Args, which would let a
|
||||
// test binary's own flags reach the command tree.
|
||||
if args == nil {
|
||||
args = []string{}
|
||||
}
|
||||
|
||||
root := newRootCommand(stderr)
|
||||
root := newRootCommand(stdout, stderr)
|
||||
root.SetArgs(args)
|
||||
|
||||
err := root.Execute()
|
||||
@@ -78,6 +92,9 @@ func run(args []string, stderr io.Writer) int {
|
||||
switch {
|
||||
case err == nil:
|
||||
return exitOK
|
||||
case errors.Is(err, errInterrupted):
|
||||
// The interrupted scan has printed its own line.
|
||||
return exitFatal
|
||||
case errors.As(err, &fatal):
|
||||
// The command ran and failed: a runtime error, reported
|
||||
// without the usage text that a usage error gets.
|
||||
@@ -85,22 +102,35 @@ func run(args []string, stderr io.Writer) int {
|
||||
|
||||
return exitFatal
|
||||
default:
|
||||
// A usage error: cobra has already printed the message and
|
||||
// the usage text.
|
||||
// A usage error, which cobra has already reported on stderr.
|
||||
return exitUsage
|
||||
}
|
||||
}
|
||||
|
||||
// newRootCommand builds the command tree. Everything on stdout is
|
||||
// machine-readable data; all human-facing output (help, usage, errors)
|
||||
// goes to stderr.
|
||||
func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||
// machine-readable data, the version line included; all human-facing
|
||||
// output (help, usage, errors) goes to stderr.
|
||||
func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
||||
var showVersion bool
|
||||
|
||||
printVersion := runE(func(context.Context, []string) error {
|
||||
_, err := fmt.Fprintf(stdout, "sfdupes %s\n", Version)
|
||||
if err != nil {
|
||||
return fmt.Errorf("write stdout: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
})
|
||||
|
||||
root := &cobra.Command{
|
||||
Use: "sfdupes",
|
||||
Short: "Find candidate duplicate files by size and head/tail SHA-256",
|
||||
Version: Version,
|
||||
Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if showVersion {
|
||||
return printVersion(cmd, args)
|
||||
}
|
||||
|
||||
// A missing subcommand prints usage and exits 2: cobra
|
||||
// prints the usage text for the returned error, and run
|
||||
// maps everything that is not a fatal error to exit 2.
|
||||
@@ -113,6 +143,11 @@ func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||
root.SetErr(stderr)
|
||||
root.CompletionOptions.DisableDefaultCmd = true
|
||||
|
||||
// Cobra's built-in version flag prints through the help writer,
|
||||
// stderr; this one prints to stdout.
|
||||
root.Flags().BoolVarP(&showVersion, "version", "v", false,
|
||||
"print the version to stdout")
|
||||
|
||||
var (
|
||||
scanWorkers int
|
||||
scanOneFS bool
|
||||
@@ -122,12 +157,18 @@ func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||
Use: cmdScan + " [--workers N] [-x] PATH...",
|
||||
Short: "Walk trees and synchronize the scan database",
|
||||
Args: cobra.MinimumNArgs(1),
|
||||
PreRunE: func(cmd *cobra.Command, _ []string) error {
|
||||
return checkScanWorkers(cmd, scanWorkers)
|
||||
},
|
||||
RunE: runE(func(ctx context.Context, args []string) error {
|
||||
ctx, stop := interruptContext(ctx)
|
||||
defer stop()
|
||||
|
||||
return runScan(ctx, args, scanWorkers, scanOneFS)
|
||||
}),
|
||||
}
|
||||
scanCmd.Flags().IntVar(&scanWorkers, "workers", runtime.NumCPU(),
|
||||
"concurrent workers for the walk and hash phases")
|
||||
"concurrent workers for the walk, hash, and content phases")
|
||||
scanCmd.Flags().BoolVarP(&scanOneFS, "one-file-system", "x", false,
|
||||
"do not cross filesystem boundaries")
|
||||
|
||||
@@ -136,7 +177,7 @@ func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||
Short: "Read the scan database and print the file-level duplicates report",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: runE(func(ctx context.Context, _ []string) error {
|
||||
return runReport(ctx)
|
||||
return runReport(ctx, stdout)
|
||||
}),
|
||||
}
|
||||
|
||||
@@ -145,7 +186,7 @@ func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||
Short: "Read the scan database and print the duplicate-tree report",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: runE(func(ctx context.Context, _ []string) error {
|
||||
return runTrees(ctx)
|
||||
return runTrees(ctx, stdout)
|
||||
}),
|
||||
}
|
||||
|
||||
@@ -154,13 +195,26 @@ func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||
return root
|
||||
}
|
||||
|
||||
// runE adapts a subcommand implementation to cobra's RunE. Cobra
|
||||
// prints the error and the command's usage text for every error RunE
|
||||
// returns, but a subcommand that ran and failed has no usage problem
|
||||
// to report: both are silenced here, and the error is marked fatal so
|
||||
// that run reports it on stderr and exits 1 rather than 2. The command's
|
||||
// context is handed to the implementation: cancelling it unwinds the
|
||||
// scan's worker pools.
|
||||
// checkScanWorkers rejects a scan --workers value below 1. That is a
|
||||
// usage error reported in one line: cobra prints the returned message
|
||||
// without the usage text, and run exits 2.
|
||||
func checkScanWorkers(cmd *cobra.Command, workers int) error {
|
||||
if workers >= 1 {
|
||||
return nil
|
||||
}
|
||||
|
||||
cmd.SilenceUsage = true
|
||||
|
||||
return fmt.Errorf("%w, got %d", errWorkersBelowOne, workers)
|
||||
}
|
||||
|
||||
// runE adapts a subcommand implementation, or the version print, to
|
||||
// cobra's RunE. Cobra prints the error and the command's usage text for
|
||||
// every error RunE returns, but a subcommand that ran and failed has no
|
||||
// usage problem to report: both are silenced here, and the error is
|
||||
// marked fatal so that run reports it on stderr and exits 1 rather than
|
||||
// 2. The command's context is handed to the implementation: cancelling
|
||||
// it unwinds the scan's worker pools.
|
||||
func runE(
|
||||
fn func(ctx context.Context, args []string) error,
|
||||
) func(*cobra.Command, []string) error {
|
||||
|
||||
+464
-69
@@ -44,23 +44,61 @@ func assertNoSidecars(t *testing.T, path string) {
|
||||
}
|
||||
}
|
||||
|
||||
// captureStdout redirects os.Stdout to a file for the rest of the test
|
||||
// and returns a function reading back everything written to it. Only
|
||||
// machine-readable data belongs on stdout (README design goal 4), so
|
||||
// the tests assert on it directly.
|
||||
func captureStdout(t *testing.T) func() string {
|
||||
// makeReadOnly takes write permission away from the database at path,
|
||||
// from any WAL sidecar beside it, and from their directory, as for a
|
||||
// user reading a database that a root cron scan keeps. Root ignores
|
||||
// file permissions, so it skips the test when run as root.
|
||||
func makeReadOnly(t *testing.T, path string) {
|
||||
t.Helper()
|
||||
|
||||
f, err := os.Create(filepath.Join(t.TempDir(), "stdout"))
|
||||
if os.Geteuid() == 0 {
|
||||
t.Skip("root ignores file permissions")
|
||||
}
|
||||
|
||||
err := os.Chmod(path, 0o400)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
saved := os.Stdout
|
||||
os.Stdout = f
|
||||
for _, suffix := range walSuffixes {
|
||||
err = os.Chmod(path+suffix, 0o400)
|
||||
if err != nil && !errors.Is(err, fs.ErrNotExist) {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
dir := filepath.Dir(path)
|
||||
|
||||
//nolint:gosec // reaching the database needs the search bit
|
||||
err = os.Chmod(dir, 0o500)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Runs before t.TempDir's own cleanup, which must delete the files.
|
||||
t.Cleanup(func() {
|
||||
//nolint:gosec // removing the directory needs its search bit back
|
||||
_ = os.Chmod(dir, 0o700)
|
||||
})
|
||||
}
|
||||
|
||||
// captureStderr redirects os.Stderr to a file for the rest of the test
|
||||
// and returns a function reading back everything written to it. scan
|
||||
// writes its warnings and summary straight to os.Stderr, not to the
|
||||
// stderr writer run is given.
|
||||
func captureStderr(t *testing.T) func() string {
|
||||
t.Helper()
|
||||
|
||||
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
saved := os.Stderr
|
||||
os.Stderr = f
|
||||
|
||||
t.Cleanup(func() {
|
||||
os.Stdout = saved
|
||||
os.Stderr = saved
|
||||
|
||||
_ = f.Close()
|
||||
})
|
||||
@@ -91,13 +129,13 @@ func captureStdout(t *testing.T) func() string {
|
||||
// brokenDatabase writes a database that opens cleanly and passes the
|
||||
// schema-version check but has no files table, so the first query
|
||||
// fails with the database already open: a fatal error on a path that
|
||||
// owns an open database.
|
||||
// owns an open database. It closes the database the way scan does.
|
||||
func brokenDatabase(t *testing.T) string {
|
||||
t.Helper()
|
||||
|
||||
path := testDBPath(t)
|
||||
|
||||
db, err := openDB(path)
|
||||
db, err := openDB(path, scanParams)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -108,10 +146,7 @@ func brokenDatabase(t *testing.T) string {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = db.Close()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
closeScanDatabase(t.Context(), db, path)
|
||||
|
||||
return path
|
||||
}
|
||||
@@ -144,7 +179,10 @@ func TestOpenDatabaseKeepsWALWhileOpen(t *testing.T) {
|
||||
|
||||
func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
|
||||
// Every subcommand that owns an open database must close it when
|
||||
// it fails: no os.Exit between the open and the return.
|
||||
// it fails: no os.Exit between the open and the return. The
|
||||
// sidecar check is evidence of the close only for scan: report and
|
||||
// trees only read a database that is out of WAL mode, which leaves
|
||||
// nothing on disk whether they close it or not.
|
||||
cases := map[string][]string{
|
||||
cmdScan: {cmdScan},
|
||||
cmdReport: {cmdReport},
|
||||
@@ -160,17 +198,15 @@ func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
|
||||
args = append(args, t.TempDir())
|
||||
}
|
||||
|
||||
var stderr bytes.Buffer
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run(args, &stderr)
|
||||
code := run(args, &stdout, &stderr)
|
||||
if code != exitFatal {
|
||||
t.Errorf("run(%v) = %d, want %d", args, code, exitFatal)
|
||||
}
|
||||
|
||||
assertNoSidecars(t, path)
|
||||
assertFatalOutput(t, stderr.String(), stdout())
|
||||
assertFatalOutput(t, stderr.String(), stdout.String())
|
||||
|
||||
// Proof that the failure happened after the open: only a
|
||||
// query against the opened database can report this.
|
||||
@@ -188,18 +224,16 @@ func TestRunMissingOperandIsFatalNotUsage(t *testing.T) {
|
||||
// must not dump the usage text.
|
||||
t.Setenv(databaseEnv, testDBPath(t))
|
||||
|
||||
var stderr bytes.Buffer
|
||||
|
||||
stdout := captureStdout(t)
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
missing := filepath.Join(t.TempDir(), "nope")
|
||||
|
||||
code := run([]string{cmdScan, missing}, &stderr)
|
||||
code := run([]string{cmdScan, missing}, &stdout, &stderr)
|
||||
if code != exitFatal {
|
||||
t.Errorf("run(scan %s) = %d, want %d", missing, code, exitFatal)
|
||||
}
|
||||
|
||||
assertFatalOutput(t, stderr.String(), stdout())
|
||||
assertFatalOutput(t, stderr.String(), stdout.String())
|
||||
}
|
||||
|
||||
// assertFatalOutput checks that a fatal error was reported the way
|
||||
@@ -243,11 +277,9 @@ func TestRunUsageErrors(t *testing.T) {
|
||||
// path that does not exist.
|
||||
t.Setenv(databaseEnv, testDBPath(t))
|
||||
|
||||
var stderr bytes.Buffer
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run(tc.args, &stderr)
|
||||
code := run(tc.args, &stdout, &stderr)
|
||||
if code != exitUsage {
|
||||
t.Errorf("run(%v) = %d, want %d", tc.args, code, exitUsage)
|
||||
}
|
||||
@@ -256,43 +288,115 @@ func TestRunUsageErrors(t *testing.T) {
|
||||
t.Errorf("stderr = %q, want %q", stderr.String(), tc.want)
|
||||
}
|
||||
|
||||
if got := stdout(); got != "" {
|
||||
if got := stdout.String(); got != "" {
|
||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestRunHelpAndVersionSucceed checks that the two informational flags
|
||||
// exit 0 and keep their human-facing output on stderr.
|
||||
//
|
||||
//nolint:paralleltest // captureStdout replaces the process-wide os.Stdout
|
||||
func TestRunHelpAndVersionSucceed(t *testing.T) {
|
||||
assertHumanOutput(t, "--help")
|
||||
assertHumanOutput(t, "--version")
|
||||
func TestRunScanRejectsWorkersBelowOne(t *testing.T) {
|
||||
// README §scan mode: --workers below 1 is a usage error reported in
|
||||
// one line on stderr, before the scan opens the database.
|
||||
for _, workers := range []string{"0", "-1"} {
|
||||
t.Run(workers, func(t *testing.T) {
|
||||
dbPath := testDBPath(t)
|
||||
t.Setenv(databaseEnv, dbPath)
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
args := []string{cmdScan, "--workers", workers, t.TempDir()}
|
||||
|
||||
code := run(args, &stdout, &stderr)
|
||||
if code != exitUsage {
|
||||
t.Errorf("run(%v) = %d, want %d", args, code, exitUsage)
|
||||
}
|
||||
|
||||
// assertHumanOutput runs sfdupes with one informational flag and checks
|
||||
// that it succeeds with its output on stderr and stdout untouched
|
||||
// (README design goal 4).
|
||||
func assertHumanOutput(t *testing.T, arg string) {
|
||||
t.Helper()
|
||||
want := "Error: --workers must be at least 1, got " + workers +
|
||||
"\n"
|
||||
if got := stderr.String(); got != want {
|
||||
t.Errorf("stderr = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
var stderr bytes.Buffer
|
||||
if got := stdout.String(); got != "" {
|
||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
||||
}
|
||||
|
||||
stdout := captureStdout(t)
|
||||
_, err := os.Stat(dbPath)
|
||||
if !errors.Is(err, fs.ErrNotExist) {
|
||||
t.Errorf("stat %s: %v, want the database never created",
|
||||
dbPath, err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
code := run([]string{arg}, &stderr)
|
||||
func TestRunHelp(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// README §Subcommands: help goes to stderr, exits 0, and leaves
|
||||
// stdout empty.
|
||||
cases := [][]string{{"--help"}, {"-h"}, {cmdScan, "--help"}}
|
||||
|
||||
for _, args := range cases {
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run(args, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Errorf("run(%v) = %d, want %d", args, code, exitOK)
|
||||
}
|
||||
|
||||
if !strings.Contains(stderr.String(), usageMarker) {
|
||||
t.Errorf("run(%v) stderr = %q, want the help text",
|
||||
args, stderr.String())
|
||||
}
|
||||
|
||||
if got := stdout.String(); got != "" {
|
||||
t.Errorf("run(%v) stdout = %q, want nothing (data only)",
|
||||
args, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunVersion(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// README §Subcommands: the version is one line on stdout, with
|
||||
// nothing on stderr, and exits 0.
|
||||
for _, arg := range []string{"--version", "-v"} {
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{arg}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Errorf("run(%s) = %d, want %d", arg, code, exitOK)
|
||||
}
|
||||
|
||||
if stderr.Len() == 0 {
|
||||
t.Errorf("run(%s) wrote nothing to stderr", arg)
|
||||
want := "sfdupes " + Version + "\n"
|
||||
if got := stdout.String(); got != want {
|
||||
t.Errorf("run(%s) stdout = %q, want %q", arg, got, want)
|
||||
}
|
||||
|
||||
if got := stdout(); got != "" {
|
||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
||||
if got := stderr.String(); got != "" {
|
||||
t.Errorf("run(%s) stderr = %q, want nothing", arg, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunVersionWriteFailureIsFatal(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// README §Error handling: a stdout write failure exits 1, reported
|
||||
// in one line on stderr.
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run([]string{"--version"}, failingWriter{}, &stderr)
|
||||
if code != exitFatal {
|
||||
t.Errorf("run(--version) = %d, want %d", code, exitFatal)
|
||||
}
|
||||
|
||||
want := "sfdupes: write stdout: " + errWriteFailed.Error() + "\n"
|
||||
if got := stderr.String(); got != want {
|
||||
t.Errorf("stderr = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -320,21 +424,31 @@ func scanFixture(t *testing.T) []string {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
var stderr bytes.Buffer
|
||||
scanOK(t, dir)
|
||||
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run([]string{cmdScan, dir}, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(scan) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
return dupes
|
||||
}
|
||||
|
||||
if got := stdout(); got != "" {
|
||||
// scanOK runs scan over operands, fails the test unless it exits 0 with
|
||||
// nothing on stdout, and returns everything it printed to stderr.
|
||||
func scanOK(t *testing.T, operands ...string) string {
|
||||
t.Helper()
|
||||
|
||||
var stdout bytes.Buffer
|
||||
|
||||
stderr := captureStderr(t)
|
||||
|
||||
code := run(append([]string{cmdScan}, operands...), &stdout, os.Stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(scan %q) = %d, want %d; stderr: %s",
|
||||
operands, code, exitOK, stderr())
|
||||
}
|
||||
|
||||
if got := stdout.String(); got != "" {
|
||||
t.Errorf("scan stdout = %q, want nothing (data only)", got)
|
||||
}
|
||||
|
||||
return dupes
|
||||
return stderr()
|
||||
}
|
||||
|
||||
func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
||||
@@ -345,24 +459,119 @@ func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
||||
assertNoSidecars(t, path)
|
||||
}
|
||||
|
||||
func TestRunScanSkipsSymlinkOperand(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dir := t.TempDir()
|
||||
writeFile(t, dir, "target/sub/f", pattern(1, 10))
|
||||
|
||||
link := filepath.Join(dir, "link")
|
||||
|
||||
err := os.Symlink(filepath.Join(dir, "target"), link)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Scanning a directory through the symlink stores a record beneath
|
||||
// the symlink's own path for a file beneath its target.
|
||||
scanOK(t, filepath.Join(link, "sub"))
|
||||
|
||||
assertOperandSkipped(t, path, link, "symlink",
|
||||
filepath.Join(link, "sub", "f"))
|
||||
}
|
||||
|
||||
func TestRunScanWalksOperandUnderSymlinkOperand(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dir := t.TempDir()
|
||||
writeFile(t, dir, "target/sub/f", pattern(1, 10))
|
||||
|
||||
link := filepath.Join(dir, "link")
|
||||
|
||||
err := os.Symlink(filepath.Join(dir, "target"), link)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// link is dropped as a symlink, but link/sub must still be scanned,
|
||||
// not dropped as lying under link.
|
||||
scanOK(t, link, filepath.Join(link, "sub"))
|
||||
|
||||
db, err := openDB(path, reportParams)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
t.Cleanup(func() { _ = db.Close() })
|
||||
|
||||
recordByPath(t, dbRecords(t, db), filepath.Join(link, "sub", "f"))
|
||||
}
|
||||
|
||||
func TestRunScanSkipsZFSOperand(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
zfs := filepath.Join(t.TempDir(), ".zfs")
|
||||
snapshot := filepath.Join(zfs, "snapshot", "hourly")
|
||||
f := writeFile(t, snapshot, "f", pattern(1, 10))
|
||||
|
||||
// An operand beneath a .zfs directory is walked, because it is not
|
||||
// itself named .zfs.
|
||||
scanOK(t, snapshot)
|
||||
|
||||
assertOperandSkipped(t, path, zfs, ".zfs directory", f)
|
||||
}
|
||||
|
||||
// assertOperandSkipped scans operand alone and checks that it is skipped
|
||||
// as kind: a warning naming it, one skip in the summary, exit 0, and the
|
||||
// record for kept, which an earlier scan stored beneath operand, still
|
||||
// in the database at dbPath.
|
||||
func assertOperandSkipped(t *testing.T, dbPath, operand, kind,
|
||||
kept string,
|
||||
) {
|
||||
t.Helper()
|
||||
|
||||
stderr := scanOK(t, operand)
|
||||
|
||||
warning := "walk " + operand + ": skipping " + kind + " operand\n"
|
||||
if !strings.Contains(stderr, warning) {
|
||||
t.Errorf("stderr = %q, want %q", stderr, warning)
|
||||
}
|
||||
|
||||
summary := "scan: 0 files seen (0 added, 0 updated, 0 removed, " +
|
||||
"0 unchanged), 1 skipped\n"
|
||||
if !strings.Contains(stderr, summary) {
|
||||
t.Errorf("stderr = %q, want %q", stderr, summary)
|
||||
}
|
||||
|
||||
db, err := openDB(dbPath, reportParams)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
t.Cleanup(func() { _ = db.Close() })
|
||||
|
||||
recordByPath(t, dbRecords(t, db), kept)
|
||||
}
|
||||
|
||||
func TestRunReportSucceeds(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dupes := scanFixture(t)
|
||||
|
||||
var stderr bytes.Buffer
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run([]string{cmdReport}, &stderr)
|
||||
code := run([]string{cmdReport}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
}
|
||||
|
||||
want := "first\tdupe\tsize\n" + dupes[0] + "\t" + dupes[1] + "\t300\n"
|
||||
if got := stdout(); got != want {
|
||||
if got := stdout.String(); got != want {
|
||||
t.Errorf("stdout = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
@@ -375,11 +584,9 @@ func TestRunTreesSucceeds(t *testing.T) {
|
||||
|
||||
dupes := scanFixture(t)
|
||||
|
||||
var stderr bytes.Buffer
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
stdout := captureStdout(t)
|
||||
|
||||
code := run([]string{cmdTrees}, &stderr)
|
||||
code := run([]string{cmdTrees}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
@@ -389,9 +596,197 @@ func TestRunTreesSucceeds(t *testing.T) {
|
||||
// trees of each other.
|
||||
want := "first\tdupe\tfiles\tsize\n" +
|
||||
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) + "\t1\t300\n"
|
||||
if got := stdout(); got != want {
|
||||
if got := stdout.String(); got != want {
|
||||
t.Errorf("stdout = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
assertNoSidecars(t, path)
|
||||
}
|
||||
|
||||
func TestRunReportsNeedOnlyReadAccess(t *testing.T) {
|
||||
// README §Database: report and trees need only read access to the
|
||||
// database file. With its directory read-only as well, SQLite
|
||||
// cannot create any file beside it.
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dupes := scanFixture(t)
|
||||
assertNoSidecars(t, path)
|
||||
makeReadOnly(t, path)
|
||||
|
||||
cases := map[string]string{
|
||||
cmdReport: "first\tdupe\tsize\n" +
|
||||
dupes[0] + "\t" + dupes[1] + "\t300\n",
|
||||
cmdTrees: "first\tdupe\tfiles\tsize\n" +
|
||||
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) +
|
||||
"\t1\t300\n",
|
||||
}
|
||||
|
||||
for name, want := range cases {
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{name}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Errorf("run(%s) = %d, want %d; stderr: %s",
|
||||
name, code, exitOK, stderr.String())
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
if got := stdout.String(); got != want {
|
||||
t.Errorf("%s stdout = %q, want %q", name, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// holdScanLock takes the lock on the database at path, as a running
|
||||
// scan does, and holds it until the test ends. It fails the test when
|
||||
// the lock is already held.
|
||||
func holdScanLock(t *testing.T, path string) {
|
||||
t.Helper()
|
||||
|
||||
lock, err := lockScanDatabase(path)
|
||||
if err != nil {
|
||||
t.Fatalf("lock %s: %v", path, err)
|
||||
}
|
||||
|
||||
t.Cleanup(func() { _ = lock.Close() })
|
||||
}
|
||||
|
||||
func TestRunSecondScanFails(t *testing.T) {
|
||||
// README §Database: while one scan holds the lock, a second scan
|
||||
// fails at once, naming the lock file, without creating the
|
||||
// database.
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
holdScanLock(t, path)
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{cmdScan, t.TempDir()}, &stdout, &stderr)
|
||||
if code != exitFatal {
|
||||
t.Errorf("run(scan) = %d, want %d", code, exitFatal)
|
||||
}
|
||||
|
||||
want := "sfdupes: another scan is running (lock held on " +
|
||||
path + ".lock)\n"
|
||||
if got := stderr.String(); got != want {
|
||||
t.Errorf("stderr = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
if got := stdout.String(); got != "" {
|
||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
||||
}
|
||||
|
||||
_, err := os.Stat(path)
|
||||
if !errors.Is(err, fs.ErrNotExist) {
|
||||
t.Errorf("stat %s = %v, want the database not created", path, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunScanReleasesLock(t *testing.T) {
|
||||
// README §Database: a scan releases the lock however it ends.
|
||||
t.Run("success", func(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
scanFixture(t)
|
||||
holdScanLock(t, path)
|
||||
})
|
||||
|
||||
t.Run("fatal error", func(t *testing.T) {
|
||||
path := brokenDatabase(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
code := run([]string{cmdScan, t.TempDir()}, io.Discard, io.Discard)
|
||||
if code != exitFatal {
|
||||
t.Fatalf("run(scan) = %d, want %d", code, exitFatal)
|
||||
}
|
||||
|
||||
holdScanLock(t, path)
|
||||
})
|
||||
}
|
||||
|
||||
func TestRunReportsDuringScan(t *testing.T) {
|
||||
// README §Database: report and trees never take the lock, so they
|
||||
// run while a scan holds it.
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
scanFixture(t)
|
||||
holdScanLock(t, path)
|
||||
|
||||
for _, name := range []string{cmdReport, cmdTrees} {
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run([]string{name}, io.Discard, &stderr)
|
||||
if code != exitOK {
|
||||
t.Errorf("run(%s) = %d, want %d; stderr: %s",
|
||||
name, code, exitOK, stderr.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunStdoutClosedIsFatal(t *testing.T) {
|
||||
// README §Error handling: a stdout write failure exits 1, reported
|
||||
// in one line on stderr.
|
||||
for _, name := range []string{cmdReport, cmdTrees} {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
t.Setenv(databaseEnv, testDBPath(t))
|
||||
|
||||
scanFixture(t)
|
||||
|
||||
stdout, err := os.Create(filepath.Join(t.TempDir(), "stdout"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = stdout.Close()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run([]string{name}, stdout, &stderr)
|
||||
if code != exitFatal {
|
||||
t.Errorf("run(%s) = %d, want %d", name, code, exitFatal)
|
||||
}
|
||||
|
||||
got := stderr.String()
|
||||
if !strings.HasPrefix(got, "sfdupes: write stdout: ") ||
|
||||
!strings.Contains(got, os.ErrClosed.Error()) ||
|
||||
strings.Count(got, "\n") != 1 {
|
||||
t.Errorf("stderr = %q, want one line reporting the "+
|
||||
"failed stdout write", got)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// errWriteFailed is the error failingWriter returns.
|
||||
var errWriteFailed = errors.New("write failed")
|
||||
|
||||
// failingWriter is a stdout that fails every write.
|
||||
type failingWriter struct{}
|
||||
|
||||
func (failingWriter) Write([]byte) (int, error) { return 0, errWriteFailed }
|
||||
|
||||
func TestStdoutWriteErrorPropagates(t *testing.T) {
|
||||
t.Setenv(databaseEnv, testDBPath(t))
|
||||
|
||||
scanFixture(t)
|
||||
|
||||
cases := map[string]func(context.Context, io.Writer) error{
|
||||
cmdReport: runReport,
|
||||
cmdTrees: runTrees,
|
||||
}
|
||||
|
||||
for name, fn := range cases {
|
||||
err := fn(t.Context(), failingWriter{})
|
||||
if !errors.Is(err, errWriteFailed) {
|
||||
t.Errorf("%s: error = %v, want %v", name, err, errWriteFailed)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+46
-17
@@ -6,6 +6,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/schollz/progressbar/v3"
|
||||
"golang.org/x/term"
|
||||
)
|
||||
|
||||
// plainInterval is the minimum time between progress lines when stderr
|
||||
@@ -24,24 +25,23 @@ const percentScale = 100
|
||||
|
||||
// stderrIsTTY reports whether stderr is attached to a terminal.
|
||||
func stderrIsTTY() bool {
|
||||
fi, err := os.Stderr.Stat()
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
|
||||
return fi.Mode()&os.ModeCharDevice != 0
|
||||
return term.IsTerminal(int(os.Stderr.Fd()))
|
||||
}
|
||||
|
||||
// progress renders one scan pass's progress on stderr. On a TTY it
|
||||
// delegates to the progressbar library (spinner style when the total is
|
||||
// unknown, full bar with count/percent/rate/elapsed/ETA otherwise). When
|
||||
// stderr is not a TTY it emits no ANSI redraws: it prints a plain
|
||||
// one-line update no more often than every plainInterval.
|
||||
// one-line update as the pass starts, then no more often than every
|
||||
// plainInterval.
|
||||
//
|
||||
// All methods must be called from the main goroutine only. A nil
|
||||
// *progress is a valid no-display receiver: every method is a no-op,
|
||||
// so batched database flushes during the streaming pass can reuse the
|
||||
// update-pass helpers without rendering anything.
|
||||
// All methods must be called from the main goroutine only. On a TTY
|
||||
// the library also redraws a spinner from its own goroutine, several
|
||||
// times a second, so its count and elapsed time stay current while a
|
||||
// pass waits for its next item. A nil *progress is a valid
|
||||
// no-display receiver: every method is a no-op, so batched database
|
||||
// flushes during the streaming pass can reuse the update-pass helpers
|
||||
// without rendering anything.
|
||||
type progress struct {
|
||||
label string
|
||||
total int64 // -1 when unknown (walk pass)
|
||||
@@ -53,10 +53,22 @@ type progress struct {
|
||||
|
||||
func newProgress(label string, total int64) *progress {
|
||||
p := &progress{label: label, total: total, start: time.Now()}
|
||||
if !stderrIsTTY() {
|
||||
if stderrIsTTY() {
|
||||
p.bar = newBar(label, total)
|
||||
|
||||
return p
|
||||
}
|
||||
|
||||
// Print the zero state at once: the first item may take minutes,
|
||||
// and a pass must never look hung.
|
||||
p.last = p.start
|
||||
fmt.Fprintln(os.Stderr, p.plainLine())
|
||||
|
||||
return p
|
||||
}
|
||||
|
||||
// newBar builds the TTY display for newProgress.
|
||||
func newBar(label string, total int64) *progressbar.ProgressBar {
|
||||
opts := []progressbar.Option{
|
||||
progressbar.OptionSetWriter(os.Stderr),
|
||||
progressbar.OptionSetDescription(label),
|
||||
@@ -81,9 +93,7 @@ func newProgress(label string, total int64) *progress {
|
||||
)
|
||||
}
|
||||
|
||||
p.bar = progressbar.NewOptions64(total, opts...)
|
||||
|
||||
return p
|
||||
return progressbar.NewOptions64(total, opts...)
|
||||
}
|
||||
|
||||
// increment records one completed item and refreshes the display.
|
||||
@@ -106,26 +116,45 @@ func (p *progress) increment() {
|
||||
}
|
||||
|
||||
// warnf prints a one-line warning to stderr without corrupting the bar.
|
||||
// The whole message is escaped like a report's path columns, so a path
|
||||
// holding a newline cannot split the warning.
|
||||
func (p *progress) warnf(format string, args ...any) {
|
||||
if p == nil {
|
||||
return
|
||||
}
|
||||
|
||||
msg := escapePath(fmt.Sprintf(format, args...))
|
||||
|
||||
if p.bar != nil && p.total < 0 {
|
||||
// The library also redraws a spinner from its own goroutine, so
|
||||
// a direct write could land inside a redraw. The bar prints the
|
||||
// warning itself, just before its next redraw.
|
||||
_, _ = progressbar.Bprintln(p.bar, msg)
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
if p.bar != nil {
|
||||
_ = p.bar.Clear()
|
||||
}
|
||||
|
||||
fmt.Fprintf(os.Stderr, format+"\n", args...)
|
||||
fmt.Fprintln(os.Stderr, msg)
|
||||
}
|
||||
|
||||
// finish terminates the pass's display.
|
||||
// finish terminates the pass's display. A bar whose pass stopped short
|
||||
// of its total, as an interrupted one does, is left as last drawn; the
|
||||
// library's Finish would fill it up.
|
||||
func (p *progress) finish() {
|
||||
if p == nil {
|
||||
return
|
||||
}
|
||||
|
||||
if p.bar != nil {
|
||||
if p.total >= 0 && p.count < p.total {
|
||||
_ = p.bar.Exit()
|
||||
} else {
|
||||
_ = p.bar.Finish()
|
||||
}
|
||||
|
||||
fmt.Fprintln(os.Stderr)
|
||||
|
||||
|
||||
@@ -0,0 +1,189 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// spinnerIdle comfortably outlasts the 100ms interval at which the
|
||||
// progressbar library redraws a spinner from its own goroutine.
|
||||
const spinnerIdle = 500 * time.Millisecond
|
||||
|
||||
//nolint:paralleltest // replaces the process-wide os.Stderr
|
||||
func TestStderrIsTTYFalseForNonTerminals(t *testing.T) {
|
||||
r, pipe, err := os.Pipe()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
regular, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
devNull, err := os.OpenFile(os.DevNull, os.O_WRONLY, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
saved := os.Stderr
|
||||
|
||||
t.Cleanup(func() {
|
||||
os.Stderr = saved
|
||||
|
||||
for _, f := range []*os.File{r, pipe, regular, devNull} {
|
||||
_ = f.Close()
|
||||
}
|
||||
})
|
||||
|
||||
cases := map[string]*os.File{
|
||||
"a pipe": pipe,
|
||||
"a regular file": regular,
|
||||
os.DevNull: devNull,
|
||||
}
|
||||
|
||||
for name, f := range cases {
|
||||
os.Stderr = f
|
||||
|
||||
if stderrIsTTY() {
|
||||
t.Errorf("stderrIsTTY() = true with stderr on %s", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestNewProgressPrintsBeforeFirstItem checks that each pass shows its
|
||||
// zero state the moment it starts when stderr is not a terminal, and
|
||||
// that the next line still waits for plainInterval.
|
||||
//
|
||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
||||
func TestNewProgressPrintsBeforeFirstItem(t *testing.T) {
|
||||
stderr := captureStderr(t)
|
||||
|
||||
newProgress("walk", -1).increment()
|
||||
newProgress("hash", 10).increment()
|
||||
|
||||
want := "walk: 0 files, elapsed 0s\n" +
|
||||
"hash: [0/10] 0% 0 files/s elapsed 0s eta ?\n"
|
||||
if got := stderr(); got != want {
|
||||
t.Errorf("stderr = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// newWalkSpinner returns the walk pass's terminal display, writing to
|
||||
// os.Stderr whether or not it is a terminal, and stops the library's
|
||||
// redraws when the test ends.
|
||||
func newWalkSpinner(t *testing.T) *progress {
|
||||
t.Helper()
|
||||
|
||||
p := &progress{
|
||||
label: "walk", total: -1, start: time.Now(),
|
||||
bar: newBar("walk", -1),
|
||||
}
|
||||
t.Cleanup(p.finish)
|
||||
|
||||
return p
|
||||
}
|
||||
|
||||
// TestProgressWarningsOnOwnLines drives the terminal display of the walk
|
||||
// pass through a run of warnings with no items between them, as when the
|
||||
// walk meets many unreadable paths, for several of the spinner's
|
||||
// redraws: every warning must land on a line of its own, never inside a
|
||||
// redraw.
|
||||
//
|
||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
||||
func TestProgressWarningsOnOwnLines(t *testing.T) {
|
||||
stderr := captureStderr(t)
|
||||
p := newWalkSpinner(t)
|
||||
|
||||
// No pause between warnings: one written straight to stderr is
|
||||
// garbled only if a redraw lands while it is being written.
|
||||
issued := 0
|
||||
for start := time.Now(); time.Since(start) < spinnerIdle; issued++ {
|
||||
p.warnf("warning")
|
||||
}
|
||||
|
||||
// The spinner prints the warnings at its next redraw.
|
||||
time.Sleep(spinnerIdle)
|
||||
|
||||
// A terminal shows each line as the text after its last carriage
|
||||
// return.
|
||||
shown := 0
|
||||
|
||||
for line := range strings.SplitSeq(stderr(), "\n") {
|
||||
if !strings.Contains(line, "warning") {
|
||||
continue
|
||||
}
|
||||
|
||||
shown++
|
||||
|
||||
if text := line[strings.LastIndex(line, "\r")+1:]; text != "warning" {
|
||||
t.Errorf("terminal shows %q, want %q", text, "warning")
|
||||
}
|
||||
}
|
||||
|
||||
if shown != issued {
|
||||
t.Errorf("%d warning lines, want %d", shown, issued)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSpinnerShowsCountAfterBurst checks that once a burst of items
|
||||
// faster than the redraw limit is over, the walk display shows every
|
||||
// item completed while it waits for the next one.
|
||||
//
|
||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
||||
func TestSpinnerShowsCountAfterBurst(t *testing.T) {
|
||||
stderr := captureStderr(t)
|
||||
p := newWalkSpinner(t)
|
||||
|
||||
for range 50 {
|
||||
p.increment()
|
||||
}
|
||||
|
||||
time.Sleep(spinnerIdle)
|
||||
|
||||
if shown := lastFrame(stderr()); !strings.Contains(shown, "(50/-,") {
|
||||
t.Errorf("terminal shows %q, want a count of 50", shown)
|
||||
}
|
||||
}
|
||||
|
||||
// lastFrame returns what a terminal shows of the frames a bar drew: the
|
||||
// last one. The library starts each frame with a carriage return and
|
||||
// erases the previous one with spaces first.
|
||||
func lastFrame(out string) string {
|
||||
var shown string
|
||||
|
||||
for frame := range strings.SplitSeq(out, "\r") {
|
||||
if strings.TrimSpace(frame) != "" {
|
||||
shown = frame
|
||||
}
|
||||
}
|
||||
|
||||
return shown
|
||||
}
|
||||
|
||||
// TestBarStoppedShortKeepsCount checks that the terminal display of a
|
||||
// pass that stops before its total, as an interrupted one does, is left
|
||||
// as last drawn instead of being filled up.
|
||||
//
|
||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
||||
func TestBarStoppedShortKeepsCount(t *testing.T) {
|
||||
stderr := captureStderr(t)
|
||||
p := &progress{
|
||||
label: "hash", total: 10, start: time.Now(),
|
||||
bar: newBar("hash", 10),
|
||||
}
|
||||
|
||||
p.increment()
|
||||
|
||||
// Past the redraw limit, so the bar draws the next count.
|
||||
time.Sleep(2 * barThrottle)
|
||||
p.increment()
|
||||
p.finish()
|
||||
|
||||
if shown := lastFrame(stderr()); !strings.Contains(shown, "(2/10,") {
|
||||
t.Errorf("terminal shows %q, want a count of 2 of 10", shown)
|
||||
}
|
||||
}
|
||||
@@ -4,8 +4,8 @@ import (
|
||||
"bufio"
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"slices"
|
||||
"strings"
|
||||
)
|
||||
|
||||
@@ -17,82 +17,70 @@ const ioBufSize = 1 << 20
|
||||
const minGroupSize = 2
|
||||
|
||||
// scanRec is one file record from the database. The signature (size,
|
||||
// head, tail) is the duplicate key; mtime is informational only and
|
||||
// used by scan for change detection.
|
||||
// head, tail, content) is the duplicate key; mtime is informational
|
||||
// only and used by scan for change detection.
|
||||
type scanRec struct {
|
||||
size int64
|
||||
mtime int64
|
||||
head string
|
||||
tail string
|
||||
content string
|
||||
path string
|
||||
}
|
||||
|
||||
// loadRecords opens the database and reads every file record for the
|
||||
// report and trees subcommands. Any database problem — including a
|
||||
// missing database — is fatal. The error is returned rather than
|
||||
// exiting, so that the deferred close — which checkpoints the SQLite
|
||||
// WAL — always runs; the database is closed before the caller formats
|
||||
// its output, so it stays closed even if that output fails.
|
||||
func loadRecords(ctx context.Context) ([]scanRec, error) {
|
||||
// runReport implements the report subcommand: it prints the file-level
|
||||
// duplicates report as TSV on stdout. SQLite groups and orders the
|
||||
// records, and each row is written as it is read, so no group is held
|
||||
// in memory. It never touches the scanned filesystem; its only I/O is
|
||||
// the database (with SQLite's temporary sort file), stdout, and stderr.
|
||||
// Any database problem, including a missing database, is fatal.
|
||||
func runReport(ctx context.Context, stdout io.Writer) error {
|
||||
dbPath := databasePath()
|
||||
|
||||
db, err := openReportDatabase(ctx, dbPath)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
return err
|
||||
}
|
||||
|
||||
defer func() { _ = db.Close() }()
|
||||
|
||||
recs, err := loadFileRows(ctx, db)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("database %s: %w", dbPath, err)
|
||||
}
|
||||
|
||||
return recs, nil
|
||||
}
|
||||
|
||||
// dupeGroup is one set of candidate-duplicate files: identical size,
|
||||
// head hash, and tail hash. paths is sorted lexicographically; the
|
||||
// first entry is the group's "first", the rest are dupes.
|
||||
type dupeGroup struct {
|
||||
size int64
|
||||
paths []string
|
||||
}
|
||||
|
||||
// runReport implements the report subcommand: it reads every record
|
||||
// from the database and prints the file-level duplicates report as TSV
|
||||
// on stdout. It never touches the scanned filesystem; its only I/O is
|
||||
// the database, stdout, and stderr.
|
||||
func runReport(ctx context.Context) error {
|
||||
recs, err := loadRecords(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
dupes := collectDupeGroups(recs)
|
||||
|
||||
out := bufio.NewWriterSize(os.Stdout, ioBufSize)
|
||||
out := bufio.NewWriterSize(stdout, ioBufSize)
|
||||
|
||||
_, err = fmt.Fprintln(out, "first\tdupe\tsize")
|
||||
if err != nil {
|
||||
return fmt.Errorf("write stdout: %w", err)
|
||||
}
|
||||
|
||||
dupeFiles := 0
|
||||
var (
|
||||
groups, dupeFiles int
|
||||
reclaimable int64
|
||||
writeErr error
|
||||
)
|
||||
|
||||
var reclaimable int64
|
||||
records, err := loadDupeRows(ctx, db,
|
||||
func(first, path string, size int64) error {
|
||||
// A group's first path is its first row; every other
|
||||
// path is a dupe.
|
||||
if path == first {
|
||||
groups++
|
||||
|
||||
for _, g := range dupes {
|
||||
for _, p := range g.paths[1:] {
|
||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
||||
g.paths[0], p, g.size)
|
||||
if err != nil {
|
||||
return fmt.Errorf("write stdout: %w", err)
|
||||
return nil
|
||||
}
|
||||
|
||||
_, writeErr = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
||||
escapePath(first), escapePath(path), size)
|
||||
dupeFiles++
|
||||
reclaimable += g.size
|
||||
reclaimable += size
|
||||
|
||||
return writeErr
|
||||
})
|
||||
|
||||
if writeErr != nil {
|
||||
return fmt.Errorf("write stdout: %w", writeErr)
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
return fmt.Errorf("database %s: %w", dbPath, err)
|
||||
}
|
||||
|
||||
err = out.Flush()
|
||||
@@ -103,54 +91,24 @@ func runReport(ctx context.Context) error {
|
||||
fmt.Fprintf(os.Stderr,
|
||||
"report: %d records read, %d duplicate groups, %d dupe files, "+
|
||||
"%s reclaimable\n",
|
||||
len(recs), len(dupes), dupeFiles, humanBytes(reclaimable))
|
||||
records, groups, dupeFiles, humanBytes(reclaimable))
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// collectDupeGroups groups records by signature and returns every group
|
||||
// with two or more paths, each group's paths sorted lexicographically,
|
||||
// groups ordered by size descending then by first path ascending.
|
||||
func collectDupeGroups(recs []scanRec) []dupeGroup {
|
||||
groups := make(map[fileSig][]string)
|
||||
|
||||
for _, r := range recs {
|
||||
// A record without hashes (its size was unique when last
|
||||
// scanned) has unknown content and is never reported as a
|
||||
// duplicate.
|
||||
if r.head == "" {
|
||||
continue
|
||||
// escapePath returns a path as it is written in a report column (README
|
||||
// "Report output format"): a backslash, tab, newline or carriage return
|
||||
// becomes \\, \t, \n or \r, and every other byte is kept as it is.
|
||||
// Grouping and sorting use the raw path, never this form.
|
||||
func escapePath(p string) string {
|
||||
// Most paths need no escaping; skip building a replacer for them.
|
||||
if !strings.ContainsAny(p, "\\\t\n\r") {
|
||||
return p
|
||||
}
|
||||
|
||||
k := fileSig{size: r.size, head: r.head, tail: r.tail}
|
||||
groups[k] = append(groups[k], r.path)
|
||||
}
|
||||
|
||||
var dupes []dupeGroup
|
||||
|
||||
for k, paths := range groups {
|
||||
if len(paths) < minGroupSize {
|
||||
continue
|
||||
}
|
||||
|
||||
slices.Sort(paths)
|
||||
dupes = append(dupes, dupeGroup{size: k.size, paths: paths})
|
||||
}
|
||||
|
||||
// Biggest reclaimable space first; ties broken by first path.
|
||||
slices.SortFunc(dupes, func(a, b dupeGroup) int {
|
||||
if a.size != b.size {
|
||||
if a.size > b.size {
|
||||
return -1
|
||||
}
|
||||
|
||||
return 1
|
||||
}
|
||||
|
||||
return strings.Compare(a.paths[0], b.paths[0])
|
||||
})
|
||||
|
||||
return dupes
|
||||
return strings.NewReplacer(
|
||||
`\`, `\\`, "\t", `\t`, "\n", `\n`, "\r", `\r`,
|
||||
).Replace(p)
|
||||
}
|
||||
|
||||
// humanBytes formats a byte count in human units (binary prefixes).
|
||||
|
||||
+297
-26
@@ -1,26 +1,269 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestCollectDupeGroups(t *testing.T) {
|
||||
// awkwardDir is a directory name holding every byte the reports escape.
|
||||
const awkwardDir = "/d/\tone\ntwo\rthree\\four"
|
||||
|
||||
// awkwardPairRecs is a duplicate pair in sibling directories /d/A and
|
||||
// awkwardDir. A raw tab sorts before "A" but its escaped form `\t`
|
||||
// sorts after it, so awkwardDir coming first shows that sorting uses
|
||||
// the raw path.
|
||||
func awkwardPairRecs() []scanRec {
|
||||
return []scanRec{
|
||||
{size: 5, head: "h", tail: "t", content: "c", path: "/d/A/f"},
|
||||
{size: 5, head: "h", tail: "t", content: "c", path: awkwardDir + "/f"},
|
||||
}
|
||||
}
|
||||
|
||||
// seedDatabase writes recs into a fresh database and returns its path.
|
||||
func seedDatabase(t *testing.T, recs []scanRec) string {
|
||||
t.Helper()
|
||||
|
||||
path := testDBPath(t)
|
||||
|
||||
db, err := openScanDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = applyChanges(t.Context(), db, recs, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = db.Close()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return path
|
||||
}
|
||||
|
||||
// dupeGroup is one duplicate group as report reads it: the size, and
|
||||
// the paths in report order, first path first.
|
||||
type dupeGroup struct {
|
||||
size int64
|
||||
paths []string
|
||||
}
|
||||
|
||||
// dupeGroups returns the duplicate groups report reads from db, in
|
||||
// report order.
|
||||
func dupeGroups(t *testing.T, db *sql.DB) []dupeGroup {
|
||||
t.Helper()
|
||||
|
||||
var groups []dupeGroup
|
||||
|
||||
_, err := loadDupeRows(t.Context(), db,
|
||||
func(first, path string, size int64) error {
|
||||
if path == first {
|
||||
groups = append(groups, dupeGroup{size: size})
|
||||
}
|
||||
|
||||
g := &groups[len(groups)-1]
|
||||
g.paths = append(g.paths, path)
|
||||
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return groups
|
||||
}
|
||||
|
||||
// dupeGroupsOf writes recs into a fresh database and returns the
|
||||
// duplicate groups report reads from it.
|
||||
func dupeGroupsOf(t *testing.T, recs []scanRec) []dupeGroup {
|
||||
t.Helper()
|
||||
|
||||
db := openTestDB(t)
|
||||
|
||||
err := applyChanges(t.Context(), db, recs, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return dupeGroups(t, db)
|
||||
}
|
||||
|
||||
func TestRunReportEscapesPaths(t *testing.T) {
|
||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{cmdReport}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
}
|
||||
|
||||
want := "first\tdupe\tsize\n" +
|
||||
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
|
||||
if got := stdout.String(); got != want {
|
||||
t.Errorf("stdout = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReportStdoutFailsWhileReading(t *testing.T) {
|
||||
// Each row holds two paths longer than dir, so the report is more
|
||||
// than twice the stdout buffer and stdout fails while rows are
|
||||
// still being read, not at the final flush.
|
||||
dir := "/" + strings.Repeat("d", 4096)
|
||||
|
||||
recs := make([]scanRec, ioBufSize/len(dir))
|
||||
for i := range recs {
|
||||
recs[i] = scanRec{
|
||||
size: 1, head: "h", tail: "t", content: "c",
|
||||
path: fmt.Sprintf("%s/%d", dir, i),
|
||||
}
|
||||
}
|
||||
|
||||
t.Setenv(databaseEnv, seedDatabase(t, recs))
|
||||
|
||||
err := runReport(t.Context(), failingWriter{})
|
||||
if !errors.Is(err, errWriteFailed) ||
|
||||
!strings.HasPrefix(err.Error(), "write stdout: ") {
|
||||
t.Errorf("error = %v, want write stdout: %v", err, errWriteFailed)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunReportsIgnoreInsertionOrder(t *testing.T) {
|
||||
// README §Constraints: identical database contents give identical
|
||||
// output, whatever order the records were inserted in.
|
||||
recs := append(smokeTreeRecs(), awkwardPairRecs()...)
|
||||
recs = append(recs,
|
||||
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/2"},
|
||||
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/1"},
|
||||
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/2"},
|
||||
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/1"},
|
||||
scanRec{size: 50, path: "/x/unhashed"},
|
||||
)
|
||||
|
||||
reversed := slices.Clone(recs)
|
||||
slices.Reverse(reversed)
|
||||
|
||||
for _, name := range []string{cmdReport, cmdTrees} {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
t.Setenv(databaseEnv, seedDatabase(t, recs))
|
||||
|
||||
forward := runStdout(t, name)
|
||||
|
||||
t.Setenv(databaseEnv, seedDatabase(t, reversed))
|
||||
|
||||
backward := runStdout(t, name)
|
||||
|
||||
if strings.Count(forward, "\n") < 3 {
|
||||
t.Errorf("stdout = %q, want at least two rows", forward)
|
||||
}
|
||||
|
||||
if forward != backward {
|
||||
t.Errorf("stdout depends on insertion order: %q vs %q",
|
||||
forward, backward)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// runStdout runs the subcommand name and returns its stdout, failing
|
||||
// the test unless it succeeds.
|
||||
func runStdout(t *testing.T, name string) string {
|
||||
t.Helper()
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{name}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(%s) = %d, want %d; stderr: %s",
|
||||
name, code, exitOK, stderr.String())
|
||||
}
|
||||
|
||||
return stdout.String()
|
||||
}
|
||||
|
||||
func TestEscapePath(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
cases := map[string]string{
|
||||
"/srv/plain": "/srv/plain",
|
||||
"/a\tb": `/a\tb`,
|
||||
"/a\nb": `/a\nb`,
|
||||
"/a\rb": `/a\rb`,
|
||||
`/a\b`: `/a\\b`,
|
||||
`/a\tb`: `/a\\tb`,
|
||||
"/not-utf8\xff": "/not-utf8\xff",
|
||||
}
|
||||
for in, want := range cases {
|
||||
if got := escapePath(in); got != want {
|
||||
t.Errorf("escapePath(%q) = %q, want %q", in, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestWarnfEscapes checks that a warning naming a path that holds a
|
||||
// newline is still one line.
|
||||
//
|
||||
//nolint:paralleltest // replaces the process-wide os.Stderr
|
||||
func TestWarnfEscapes(t *testing.T) {
|
||||
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
saved := os.Stderr
|
||||
os.Stderr = f
|
||||
|
||||
t.Cleanup(func() {
|
||||
os.Stderr = saved
|
||||
|
||||
_ = f.Close()
|
||||
})
|
||||
|
||||
(&progress{}).warnf("stat %s: %s", "/d/a\nb", "gone")
|
||||
|
||||
_, err = f.Seek(0, io.SeekStart)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
got, err := io.ReadAll(f)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
want := `stat /d/a\nb: gone` + "\n"
|
||||
if string(got) != want {
|
||||
t.Errorf("warning = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDupeGroups(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
recs := []scanRec{
|
||||
{size: 100, head: "h", tail: "t", path: "/z/b"},
|
||||
{size: 100, head: "h", tail: "t", path: "/z/a"},
|
||||
{size: 100, head: "h", tail: "t", path: "/z/c"},
|
||||
{size: 4000, head: "H", tail: "T", path: "/big/2"},
|
||||
{size: 4000, head: "H", tail: "T", path: "/big/1"},
|
||||
{size: 100, head: "h", tail: "t", content: "c", path: "/z/b"},
|
||||
{size: 100, head: "h", tail: "t", content: "c", path: "/z/a"},
|
||||
{size: 100, head: "h", tail: "t", content: "c", path: "/z/c"},
|
||||
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/2"},
|
||||
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/1"},
|
||||
// Same size as the /z group but a different head hash.
|
||||
{size: 100, head: "other", tail: "t", path: "/z/d"},
|
||||
{size: 100, head: "other", tail: "t", content: "c", path: "/z/d"},
|
||||
// A singleton signature must not form a group.
|
||||
{size: 7, head: "u", tail: "u", path: "/lonely"},
|
||||
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
groups := dupeGroupsOf(t, recs)
|
||||
if len(groups) != 2 {
|
||||
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
||||
}
|
||||
@@ -38,33 +281,61 @@ func TestCollectDupeGroups(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
|
||||
func TestDupeGroupsContentSeparates(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// Same size, head, and tail, but different content hashes: the final
|
||||
// rung keeps them apart, so no group forms. Matching content groups.
|
||||
// Records without a content hash never group, not even with each
|
||||
// other.
|
||||
recs := []scanRec{
|
||||
{size: 100, head: "h", tail: "t", content: "c1", path: "/a"},
|
||||
{size: 100, head: "h", tail: "t", content: "c2", path: "/b"},
|
||||
{size: 100, head: "h", tail: "t", content: "c1", path: "/c"},
|
||||
{size: 100, head: "h", tail: "t", path: "/d"},
|
||||
{size: 100, head: "h", tail: "t", path: "/e"},
|
||||
}
|
||||
|
||||
groups := dupeGroupsOf(t, recs)
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
||||
len(groups))
|
||||
}
|
||||
|
||||
if !slices.Equal(groups[0].paths, []string{"/a", "/c"}) {
|
||||
t.Errorf("group paths = %q, want /a /c", groups[0].paths)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDupeGroupsMtimeExcluded(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// mtime is informational only; records differing only in mtime
|
||||
// still group together.
|
||||
recs := []scanRec{
|
||||
{size: 9, mtime: 100, head: "h", tail: "t", path: "/m/1"},
|
||||
{size: 9, mtime: 200, head: "h", tail: "t", path: "/m/2"},
|
||||
{size: 9, mtime: 100, head: "h", tail: "t", content: "c", path: "/m/1"},
|
||||
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
groups := dupeGroupsOf(t, recs)
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("len(groups) = %d, want 1", len(groups))
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
||||
func TestDupeGroupsTieBreak(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// The hashes sort opposite to the first paths, so ordering the
|
||||
// groups by hash instead of by first path fails this test.
|
||||
recs := []scanRec{
|
||||
{size: 50, head: "b", tail: "b", path: "/beta/2"},
|
||||
{size: 50, head: "b", tail: "b", path: "/beta/1"},
|
||||
{size: 50, head: "a", tail: "a", path: "/alpha/2"},
|
||||
{size: 50, head: "a", tail: "a", path: "/alpha/1"},
|
||||
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/2"},
|
||||
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/1"},
|
||||
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/2"},
|
||||
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/1"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
groups := dupeGroupsOf(t, recs)
|
||||
if len(groups) != 2 {
|
||||
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
||||
}
|
||||
@@ -76,22 +347,22 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroupsDeterministic(t *testing.T) {
|
||||
func TestDupeGroupsDeterministic(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
recs := []scanRec{
|
||||
{size: 1, head: "a", tail: "a", path: "/p/1"},
|
||||
{size: 1, head: "a", tail: "a", path: "/p/2"},
|
||||
{size: 2, head: "b", tail: "b", path: "/q/1"},
|
||||
{size: 2, head: "b", tail: "b", path: "/q/2"},
|
||||
{size: 1, head: "a", tail: "a", content: "a", path: "/p/1"},
|
||||
{size: 1, head: "a", tail: "a", content: "a", path: "/p/2"},
|
||||
{size: 2, head: "b", tail: "b", content: "b", path: "/q/1"},
|
||||
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
||||
}
|
||||
|
||||
forward := collectDupeGroups(recs)
|
||||
forward := dupeGroupsOf(t, recs)
|
||||
|
||||
reversed := slices.Clone(recs)
|
||||
slices.Reverse(reversed)
|
||||
|
||||
backward := collectDupeGroups(reversed)
|
||||
backward := dupeGroupsOf(t, reversed)
|
||||
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
|
||||
return a.size == b.size && slices.Equal(a.paths, b.paths)
|
||||
}) {
|
||||
|
||||
@@ -6,9 +6,12 @@ import (
|
||||
"crypto/sha256"
|
||||
"database/sql"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"io/fs"
|
||||
"os"
|
||||
"os/signal"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strings"
|
||||
@@ -16,13 +19,49 @@ import (
|
||||
"syscall"
|
||||
)
|
||||
|
||||
// chunk is the number of bytes hashed from each end of a file.
|
||||
const chunk = 1024
|
||||
// The duplicate ladder (see hashSignature and README "Duplicate
|
||||
// detection"). A same-size candidate below headTailMin is hashed in
|
||||
// full and compared directly; a larger one is separated first by the
|
||||
// hashes of its end windows, then by a content hash that is exact below
|
||||
// wholeFileMax and deliberately sampled at or above it. The hash phase
|
||||
// reads only the end windows of a larger file; the content phase reads
|
||||
// it for its content hash only once its size, head, and tail match
|
||||
// another file's.
|
||||
|
||||
// headTailMin is the size threshold for the end-window gate. A file
|
||||
// smaller than this is hashed in full directly, with no separate head
|
||||
// and tail step: its head, tail, and content all carry the whole-file
|
||||
// hash. A file this size or larger is separated first by its end
|
||||
// windows.
|
||||
const headTailMin = 10 * 1024 * 1024
|
||||
|
||||
// headTailWindow is the number of bytes hashed from each end of a file
|
||||
// at or above headTailMin (the head and tail rungs). Because
|
||||
// headTailMin is far larger than two windows, the head and tail windows
|
||||
// never overlap.
|
||||
const headTailWindow = 64 * 1024
|
||||
|
||||
// wholeFileMax is the size boundary between the two content rungs: a
|
||||
// file strictly smaller than this is content-hashed in full; a file
|
||||
// this size or larger is content-hashed by sampling.
|
||||
const wholeFileMax = 50 * 1024 * 1024
|
||||
|
||||
// sampleStride is the spacing between content samples for large files:
|
||||
// one window is read at each gigabyte-aligned offset (0, 1 GiB, ...).
|
||||
const sampleStride = 1024 * 1024 * 1024
|
||||
|
||||
// sampleWindow is the number of bytes read at each large-file sample
|
||||
// offset, truncated at end of file.
|
||||
const sampleWindow = 1024 * 1024
|
||||
|
||||
// workQueueDepth bounds the job and result channels feeding the walk
|
||||
// and hash worker pools.
|
||||
const workQueueDepth = 1024
|
||||
|
||||
// errInterrupted reports a scan stopped by SIGINT or SIGTERM. runScan
|
||||
// has already printed its line, so run prints nothing more.
|
||||
var errInterrupted = errors.New("scan interrupted")
|
||||
|
||||
// fileRec carries one statted file between the scan phases. dev and
|
||||
// ino identify the underlying inode so hard-linked paths can share
|
||||
// one read; both are zero when the platform exposes no inode.
|
||||
@@ -44,23 +83,26 @@ type fileMeta struct {
|
||||
hashed bool
|
||||
}
|
||||
|
||||
// runScan implements the scan subcommand: three sequential phases —
|
||||
// walk (which stats each file as it is discovered), hash, update —
|
||||
// that synchronize the persistent database with the filesystem state
|
||||
// under the PATH operands. Only files whose size at least one other
|
||||
// file shares are ever hashed: a size-unique file cannot be a
|
||||
// duplicate. Flag parsing and the at-least-one-operand check are done
|
||||
// by cobra. Errors are returned rather than exiting, so that the
|
||||
// deferred close — which checkpoints the SQLite WAL — always runs.
|
||||
// Cancelling ctx unwinds the worker pools and aborts the scan with the
|
||||
// context's error.
|
||||
// runScan implements the scan subcommand: four sequential phases —
|
||||
// walk (which stats each file as it is discovered), hash, update,
|
||||
// content — that synchronize the persistent database with the
|
||||
// filesystem state under the PATH operands. Only files whose size at
|
||||
// least one other file shares are ever hashed: a size-unique file
|
||||
// cannot be a duplicate. A file of headTailMin or more gets its content
|
||||
// hash only when its size, head, and tail match another file's. Flag
|
||||
// parsing and the at-least-one-operand check are done by cobra. The
|
||||
// scan holds the lock on the database for its whole run, so a second
|
||||
// scan fails before it walks the filesystem or opens the database.
|
||||
// Errors are returned rather than exiting, so that the deferred close —
|
||||
// which takes the database out of WAL mode — always runs, and the lock
|
||||
// is released after it. When ctx is cancelled, as by the SIGINT or
|
||||
// SIGTERM that interruptContext catches, the scan keeps what it has
|
||||
// hashed (see syncScan), prints how many files its walk reached, and
|
||||
// returns errInterrupted. workers must be at least 1; the scan command
|
||||
// rejects anything less.
|
||||
func runScan(ctx context.Context, roots []string, workers int,
|
||||
oneFS bool,
|
||||
) error {
|
||||
if workers < 1 {
|
||||
workers = 1
|
||||
}
|
||||
|
||||
roots, err := resolveRoots(roots)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -68,14 +110,31 @@ func runScan(ctx context.Context, roots []string, workers int,
|
||||
|
||||
dbPath := databasePath()
|
||||
|
||||
db, err := openScanDatabase(ctx, dbPath)
|
||||
lock, err := lockScanDatabase(dbPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
defer func() { _ = db.Close() }()
|
||||
defer func() { _ = lock.Close() }()
|
||||
|
||||
db, err := openScanDatabase(ctx, dbPath)
|
||||
if err != nil && ctx.Err() != nil {
|
||||
// Interrupted while opening; SQLite may report that with an
|
||||
// error of its own rather than the context's.
|
||||
return interrupted(0)
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
defer closeScanDatabase(ctx, db, dbPath)
|
||||
|
||||
st, err := syncScan(ctx, db, roots, workers, oneFS)
|
||||
if errors.Is(err, context.Canceled) {
|
||||
return interrupted(st.walked)
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
return fmt.Errorf("update database %s: %w", dbPath, err)
|
||||
}
|
||||
@@ -89,6 +148,34 @@ func runScan(ctx context.Context, roots []string, workers int,
|
||||
return nil
|
||||
}
|
||||
|
||||
// interruptContext returns a copy of ctx that the first SIGINT or
|
||||
// SIGTERM cancels; the scan command runs the scan under it. stop
|
||||
// releases the signals.
|
||||
func interruptContext(ctx context.Context) (context.Context, func()) {
|
||||
// A SIGINT ignored from the start, as by a script's background job,
|
||||
// stays ignored.
|
||||
signals := []os.Signal{syscall.SIGTERM}
|
||||
if !signal.Ignored(syscall.SIGINT) {
|
||||
signals = append(signals, syscall.SIGINT)
|
||||
}
|
||||
|
||||
ctx, stop := signal.NotifyContext(ctx, signals...)
|
||||
|
||||
// Stopping restores the default handling, so a second signal ends
|
||||
// the process at once.
|
||||
context.AfterFunc(ctx, stop)
|
||||
|
||||
return ctx, stop
|
||||
}
|
||||
|
||||
// interrupted prints the line for a scan stopped by a signal after its
|
||||
// walk reached walked files, and returns errInterrupted.
|
||||
func interrupted(walked int) error {
|
||||
fmt.Fprintf(os.Stderr, "scan: interrupted after %d files\n", walked)
|
||||
|
||||
return errInterrupted
|
||||
}
|
||||
|
||||
// resolveRoots converts each PATH operand to an absolute, lexically
|
||||
// cleaned path (symlinks are not resolved) and verifies that it
|
||||
// exists. Database records are keyed by absolute path, so scan results
|
||||
@@ -142,8 +229,9 @@ func pruneRoots(roots []string) []string {
|
||||
}
|
||||
|
||||
// scanStats summarizes one scan's database synchronization for the
|
||||
// final stderr summary.
|
||||
// final stderr summary, or for the line an interrupted scan prints.
|
||||
type scanStats struct {
|
||||
walked int // files the walk reached
|
||||
added int
|
||||
updated int
|
||||
removed int
|
||||
@@ -165,49 +253,106 @@ type scanState struct {
|
||||
st scanStats
|
||||
}
|
||||
|
||||
// syncScan synchronizes the database with the filesystem under roots
|
||||
// in three sequential phases: walk (enumerate and stat every file,
|
||||
// building a complete size census), hash (read only the new or
|
||||
// changed — or previously unhashed — files whose size at least one
|
||||
// other file shares, committing results in batches as they arrive),
|
||||
// and update (record the size-unique files without reading them, and
|
||||
// delete the records the scan no longer verifies). Records outside
|
||||
// the roots are never touched.
|
||||
// syncScan synchronizes the database with the filesystem under roots;
|
||||
// see runPhases. When ctx is cancelled, as by an interrupt, it commits
|
||||
// the hashed records still waiting in the batch, starts no other write
|
||||
// or deletion, and returns the cancellation.
|
||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||
workers int, oneFS bool,
|
||||
) (scanStats, error) {
|
||||
roots = pruneRoots(roots)
|
||||
|
||||
s := &scanState{db: db}
|
||||
|
||||
err := s.runPhases(ctx, roots, workers, oneFS)
|
||||
if err == nil || ctx.Err() == nil {
|
||||
return s.st, err
|
||||
}
|
||||
|
||||
// The one write made after the cancellation, so it cannot use ctx.
|
||||
err = applyChanges(context.WithoutCancel(ctx), db, s.batch, nil, nil)
|
||||
if err != nil {
|
||||
return s.st, err
|
||||
}
|
||||
|
||||
return s.st, ctx.Err()
|
||||
}
|
||||
|
||||
// runPhases synchronizes the database with the filesystem under roots
|
||||
// in four sequential phases: walk (enumerate and stat every file,
|
||||
// building a complete size census), hash (read only the new or
|
||||
// changed — or previously unhashed — files whose size at least one
|
||||
// other file shares, committing results in batches as they arrive),
|
||||
// update (record the size-unique files without reading them, and
|
||||
// delete the records the scan no longer verifies), and content (fill
|
||||
// in the content hash of every record of headTailMin or more whose
|
||||
// size, head, and tail match another record's). Records outside the
|
||||
// roots are never touched, except that the content phase fills in
|
||||
// their content hash. Operands the walk cannot start from are dropped
|
||||
// first, so the records beneath them count as outside the roots unless
|
||||
// they lie under another root.
|
||||
func (s *scanState) runPhases(ctx context.Context, roots []string,
|
||||
workers int, oneFS bool,
|
||||
) error {
|
||||
// Types are checked before pruning so that an operand under a
|
||||
// dropped one is still scanned, not dropped as lying under it.
|
||||
roots = pruneRoots(s.walkableRoots(roots))
|
||||
|
||||
err := s.loadIndex(ctx, roots)
|
||||
if err != nil {
|
||||
return s.st, err
|
||||
return err
|
||||
}
|
||||
|
||||
changed, unhashed := s.walkPhase(startWalk(ctx, roots, oneFS, workers))
|
||||
|
||||
// A cancelled walk stops early, so its size census covers only part
|
||||
// of the roots, and every file it never reached looks vanished to
|
||||
// the update phase. Defence in depth rather than the only barrier:
|
||||
// that phase would today fail on its first BeginTx with the same
|
||||
// cancelled context before deleting anything. But it is the barrier
|
||||
// that survives a later decision to let an interrupted scan commit
|
||||
// what it has, and it turns a confusing failure deep in the update
|
||||
// phase into a clean abort at the phase boundary.
|
||||
// of the roots, and every file it never reached would look vanished
|
||||
// to the update phase. Stop before anything is written or deleted.
|
||||
err = ctx.Err()
|
||||
if err != nil {
|
||||
return s.st, err
|
||||
return err
|
||||
}
|
||||
|
||||
s.partition(changed, unhashed)
|
||||
|
||||
err = s.hashPhase(ctx, workers)
|
||||
if err != nil {
|
||||
return s.st, err
|
||||
return err
|
||||
}
|
||||
|
||||
return s.st, s.updatePhase(ctx)
|
||||
err = s.updatePhase(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
return s.contentPhase(ctx, workers)
|
||||
}
|
||||
|
||||
// walkableRoots returns the operands the walk can start from: regular
|
||||
// files, and directories not named .zfs. Every other operand is warned
|
||||
// about, counted as skipped, and dropped. A dropped operand is no
|
||||
// longer a root, so the records stored beneath it count as outside the
|
||||
// roots and are not deleted as unverified, unless it lies under another
|
||||
// root. An operand that fails lstat here is kept, and the walk warns
|
||||
// about it.
|
||||
func (s *scanState) walkableRoots(roots []string) []string {
|
||||
kept := make([]string, 0, len(roots))
|
||||
|
||||
for _, root := range roots {
|
||||
fi, err := os.Lstat(root)
|
||||
if err == nil {
|
||||
warn := operandWarning(root, fi)
|
||||
if warn != "" {
|
||||
s.st.skipped++
|
||||
|
||||
fmt.Fprintln(os.Stderr, escapePath(warn))
|
||||
|
||||
continue
|
||||
}
|
||||
}
|
||||
|
||||
kept = append(kept, root)
|
||||
}
|
||||
|
||||
return kept
|
||||
}
|
||||
|
||||
// loadIndex indexes the database records under the scan roots for
|
||||
@@ -241,7 +386,8 @@ func (s *scanState) loadIndex(ctx context.Context, roots []string) error {
|
||||
|
||||
// walkPhase drains the walk, appending every walked file's size to
|
||||
// the census and resolving what it can immediately: an unchanged file
|
||||
// whose record already has hashes needs nothing further. It returns
|
||||
// whose record already has hashes needs nothing from the hash phase
|
||||
// (the content phase may still fill in its content hash). It returns
|
||||
// the new-or-changed files and the unchanged files whose records lack
|
||||
// hashes; both remain candidates until the census decides whether
|
||||
// their sizes are shared.
|
||||
@@ -262,6 +408,7 @@ func (s *scanState) walkPhase(
|
||||
}
|
||||
|
||||
s.sizes = append(s.sizes, ev.rec.size)
|
||||
s.st.walked++
|
||||
|
||||
prog.increment()
|
||||
|
||||
@@ -380,27 +527,40 @@ func sameInode(a, b fileRec) bool {
|
||||
return (a.dev != 0 || a.ino != 0) && a.dev == b.dev && a.ino == b.ino
|
||||
}
|
||||
|
||||
// hashPhase hashes every queued file with the worker pool — one read
|
||||
// per inode run, in inode order — committing completed records to the
|
||||
// database in batches as results arrive, so a long scan persists its
|
||||
// progress as it goes (an interrupted scan resumes cheaply: the next
|
||||
// run skips everything already recorded). The total counts actual
|
||||
// reads, so the bar shows a real ETA. A run that fails to hash is
|
||||
// warned about and skipped; stale records for its paths, if any, are
|
||||
// deleted by the update phase.
|
||||
// hashPhase hashes every queued file with hashSignature — the head and
|
||||
// tail of a file of headTailMin or more, the whole file below that —
|
||||
// committing completed records to the database in batches as results
|
||||
// arrive, so a long scan persists its progress as it goes (an
|
||||
// interrupted scan resumes cheaply: the next run skips everything
|
||||
// already recorded). A run that fails to hash is warned about and
|
||||
// skipped; stale records for its paths, if any, are deleted by the
|
||||
// update phase.
|
||||
func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
||||
runs := hashRuns(s.toHash)
|
||||
s.toHash = nil
|
||||
|
||||
return s.readRuns(ctx, workers, "hash", runs, hashSignature, s.recordRun)
|
||||
}
|
||||
|
||||
// readRuns reads runs with the worker pool, one read per inode run, in
|
||||
// the order given, under a progress display named label. The workers
|
||||
// compute each run's hashes with hash, and each result goes to record;
|
||||
// a run that fails to read is warned about and counted as skipped
|
||||
// instead. The total counts actual reads, so the bar shows a real ETA.
|
||||
//
|
||||
// Returning early — a failed database write, or a cancelled scan — must
|
||||
// not strand the pool: the feeder would park forever on a full jobs
|
||||
// channel and every worker on a full results channel. The deferred stop
|
||||
// is what prevents that.
|
||||
func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
||||
runs := hashRuns(s.toHash)
|
||||
s.toHash = nil
|
||||
|
||||
pool := startHashPool(ctx, runs, workers)
|
||||
func (s *scanState) readRuns(ctx context.Context, workers int,
|
||||
label string, runs [][]fileRec,
|
||||
hash func(path string, size int64) (string, string, string, error),
|
||||
record func(ctx context.Context, r hashResult) error,
|
||||
) error {
|
||||
pool := startHashPool(ctx, runs, workers, hash)
|
||||
defer pool.stop()
|
||||
|
||||
prog := newProgress("hash", int64(len(runs)))
|
||||
prog := newProgress(label, int64(len(runs)))
|
||||
defer prog.finish()
|
||||
|
||||
for range runs {
|
||||
@@ -417,12 +577,12 @@ func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
||||
if r.err != nil {
|
||||
s.st.skipped += len(r.run)
|
||||
|
||||
prog.warnf("hash %s: %v", r.run[0].path, r.err)
|
||||
prog.warnf("%s %s: %v", label, r.run[0].path, r.err)
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
err := s.recordRun(ctx, r)
|
||||
err := record(ctx, r)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -443,18 +603,31 @@ func (s *scanState) recordRun(ctx context.Context, r hashResult) error {
|
||||
mtime: rec.mtime,
|
||||
head: r.head,
|
||||
tail: r.tail,
|
||||
content: r.content,
|
||||
path: rec.path,
|
||||
})
|
||||
}
|
||||
|
||||
return s.commitFullBatch(ctx)
|
||||
}
|
||||
|
||||
// commitFullBatch commits the running batch once it holds
|
||||
// updateBatchSize records. A batch that fails to commit is kept: the
|
||||
// commit fails when the scan is interrupted, and syncScan then commits
|
||||
// the batch itself.
|
||||
func (s *scanState) commitFullBatch(ctx context.Context) error {
|
||||
if len(s.batch) < updateBatchSize {
|
||||
return nil
|
||||
}
|
||||
|
||||
err := applyBatch(ctx, s.db, s.batch, nil, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
s.batch = s.batch[:0]
|
||||
|
||||
return err
|
||||
return nil
|
||||
}
|
||||
|
||||
// updatePhase writes the scan's tail under one progress display: the
|
||||
@@ -504,6 +677,147 @@ func (s *scanState) updatePhase(ctx context.Context) error {
|
||||
return applyChanges(ctx, s.db, nil, deletes, prog)
|
||||
}
|
||||
|
||||
// contentPhase fills in the content hash of every record of headTailMin
|
||||
// or more that lacks one and whose size, head, and tail equal another
|
||||
// record's, anywhere in the database: records from this scan and
|
||||
// records stored by earlier scans, inside or outside the roots. Only
|
||||
// such a file can still be a duplicate, so no other file of headTailMin
|
||||
// or more is read beyond its end windows. The files are read with the
|
||||
// hash phase's worker pool and their records written back in batches. A
|
||||
// failed read is warned about and counted as skipped; the record keeps
|
||||
// its empty content, so it is never grouped, and a later scan tries
|
||||
// again.
|
||||
func (s *scanState) contentPhase(ctx context.Context, workers int) error {
|
||||
toRead, recs, err := s.contentCandidates(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
err = s.readRuns(ctx, workers, "content", hashRuns(toRead),
|
||||
hashContentOnly, func(ctx context.Context, r hashResult) error {
|
||||
// Every path in the run keeps its record's head and tail
|
||||
// and gains the one content hash read for the run.
|
||||
for _, f := range r.run {
|
||||
rec := recs[f.path]
|
||||
rec.content = r.content
|
||||
s.batch = append(s.batch, rec)
|
||||
}
|
||||
|
||||
return s.commitFullBatch(ctx)
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
return applyChanges(ctx, s.db, s.batch, nil, nil)
|
||||
}
|
||||
|
||||
// contentCandidates returns the files the content phase reads, and
|
||||
// their records by path. Every record contentCandidatesSQL returns has
|
||||
// its file checked with lstat, whether or not it already has a content
|
||||
// hash: a file that is gone, is no longer a regular file, or has
|
||||
// changed by the walk's rule keeps its record as it is and does not
|
||||
// count as a match for the others, and any other lstat error is warned
|
||||
// about and counted as skipped, with the same result. If such a record
|
||||
// has no content hash, it stays out of duplicate groups; if it has one,
|
||||
// it is still reported until a scan covering its own tree updates or
|
||||
// removes it. The files of a group that pass and have no content hash
|
||||
// are read only if at least minGroupSize of the group's files pass, so
|
||||
// a group whose other members are all stale costs no reads. Only the
|
||||
// records to be read are kept.
|
||||
func (s *scanState) contentCandidates(
|
||||
ctx context.Context,
|
||||
) ([]fileRec, map[string]scanRec, error) {
|
||||
// The query and the checks take real time on a large database;
|
||||
// without a display the scan looks hung before the reads begin.
|
||||
prog := newProgress("content", -1)
|
||||
defer prog.finish()
|
||||
|
||||
var (
|
||||
toRead []fileRec
|
||||
first scanRec // the current group's first record
|
||||
passed int // the current group's files that passed the check
|
||||
unread []fileRec // those of them without a content hash
|
||||
)
|
||||
|
||||
recs := make(map[string]scanRec)
|
||||
|
||||
// endGroup queues the current group's files to read if at least
|
||||
// minGroupSize of its files passed, and drops their records if not.
|
||||
endGroup := func() {
|
||||
if passed >= minGroupSize {
|
||||
toRead = append(toRead, unread...)
|
||||
} else {
|
||||
for _, f := range unread {
|
||||
delete(recs, f.path)
|
||||
}
|
||||
}
|
||||
|
||||
passed, unread = 0, nil
|
||||
}
|
||||
|
||||
err := loadContentCandidates(ctx, s.db, func(r scanRec, hashed bool) {
|
||||
prog.increment()
|
||||
|
||||
if r.size != first.size || r.head != first.head || r.tail != first.tail {
|
||||
endGroup()
|
||||
|
||||
first = r
|
||||
}
|
||||
|
||||
f, ok, err := unchangedFile(r)
|
||||
if err != nil {
|
||||
s.st.skipped++
|
||||
|
||||
prog.warnf("content %s: %v", r.path, err)
|
||||
}
|
||||
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
|
||||
passed++
|
||||
|
||||
if !hashed {
|
||||
unread = append(unread, f)
|
||||
recs[r.path] = r
|
||||
}
|
||||
})
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
|
||||
endGroup()
|
||||
|
||||
return toRead, recs, nil
|
||||
}
|
||||
|
||||
// unchangedFile lstats the file r names and returns it for reading if
|
||||
// it is still the regular file r records: the same size, and an mtime
|
||||
// no newer than recorded (the walk's change rule). A file that is gone
|
||||
// or has changed reports false; any other lstat error is returned.
|
||||
func unchangedFile(r scanRec) (fileRec, bool, error) {
|
||||
fi, err := os.Lstat(r.path)
|
||||
if errors.Is(err, fs.ErrNotExist) {
|
||||
return fileRec{}, false, nil
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
return fileRec{}, false, err
|
||||
}
|
||||
|
||||
if !fi.Mode().IsRegular() || fi.Size() != r.size ||
|
||||
fi.ModTime().Unix() > r.mtime {
|
||||
return fileRec{}, false, nil
|
||||
}
|
||||
|
||||
dev, ino := inodeOfInfo(fi)
|
||||
|
||||
return fileRec{
|
||||
path: r.path, size: r.size, mtime: r.mtime, dev: dev, ino: ino,
|
||||
}, true, nil
|
||||
}
|
||||
|
||||
// underAnyRoot reports whether path is any of the roots or lies under
|
||||
// one of them.
|
||||
func underAnyRoot(path string, roots []string) bool {
|
||||
@@ -584,11 +898,43 @@ func sendEvent(ctx context.Context, events chan<- walkEvent,
|
||||
}
|
||||
}
|
||||
|
||||
// operandWarning returns the one-line warning for an operand the walk
|
||||
// does not start from, naming the path and what it is, or "" for one it
|
||||
// does: a regular file, or a directory not named .zfs. Symlinks are
|
||||
// never followed, including as operands.
|
||||
func operandWarning(root string, fi fs.FileInfo) string {
|
||||
var kind string
|
||||
|
||||
switch mode := fi.Mode(); {
|
||||
case mode.IsRegular():
|
||||
return ""
|
||||
case mode.IsDir():
|
||||
if filepath.Base(root) != ".zfs" {
|
||||
return ""
|
||||
}
|
||||
|
||||
kind = ".zfs directory"
|
||||
case mode&fs.ModeSymlink != 0:
|
||||
kind = "symlink"
|
||||
case mode&fs.ModeSocket != 0:
|
||||
kind = "socket"
|
||||
case mode&fs.ModeNamedPipe != 0:
|
||||
kind = "FIFO"
|
||||
case mode&fs.ModeDevice != 0:
|
||||
kind = "device node"
|
||||
default:
|
||||
kind = "non-regular file"
|
||||
}
|
||||
|
||||
return fmt.Sprintf("walk %s: skipping %s operand", root, kind)
|
||||
}
|
||||
|
||||
// seedRoot turns one PATH operand into the walk's starting state: a
|
||||
// regular-file operand is statted and emitted directly, a directory
|
||||
// operand becomes an initial job, and a symlink or other non-regular
|
||||
// operand yields nothing (symlinks are never followed, including as
|
||||
// operands).
|
||||
// regular-file operand is statted and emitted directly, and a directory
|
||||
// operand becomes an initial job. walkableRoots has already dropped
|
||||
// every other operand. One that has changed into something else since
|
||||
// is warned about and skipped here; it is still a root, so the records
|
||||
// stored beneath it are deleted as unverified.
|
||||
func seedRoot(ctx context.Context, root string,
|
||||
events chan<- walkEvent,
|
||||
) []dirJob {
|
||||
@@ -602,16 +948,19 @@ func seedRoot(ctx context.Context, root string,
|
||||
return nil
|
||||
}
|
||||
|
||||
switch {
|
||||
case fi.IsDir():
|
||||
if filepath.Base(root) == ".zfs" {
|
||||
warn := operandWarning(root, fi)
|
||||
if warn != "" {
|
||||
sendEvent(ctx, events, walkEvent{warn: warn, fail: true})
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
if fi.IsDir() {
|
||||
dev, ok := deviceOfInfo(fi)
|
||||
|
||||
return []dirJob{{path: root, rootDev: dev, rootDevOK: ok}}
|
||||
case fi.Mode().IsRegular():
|
||||
}
|
||||
|
||||
dev, ino := inodeOfInfo(fi)
|
||||
|
||||
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
||||
@@ -623,9 +972,6 @@ func seedRoot(ctx context.Context, root string,
|
||||
}})
|
||||
|
||||
return nil
|
||||
default:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
// startWalkWorkers starts the walk worker pool. Each worker processes
|
||||
@@ -724,6 +1070,12 @@ func walkOneDir(ctx context.Context, job dirJob, oneFS bool,
|
||||
var subs []dirJob
|
||||
|
||||
for _, e := range entries {
|
||||
// A cancelled scan wants nothing more from this directory: stop
|
||||
// rather than lstat the rest of a large one.
|
||||
if ctx.Err() != nil {
|
||||
return nil
|
||||
}
|
||||
|
||||
p := filepath.Join(job.path, e.Name())
|
||||
|
||||
if e.IsDir() {
|
||||
@@ -834,12 +1186,15 @@ func inodeOfInfo(fi fs.FileInfo) (uint64, uint64) {
|
||||
return statDev(st), st.Ino
|
||||
}
|
||||
|
||||
// hashResult carries one inode run's head/tail hashes (or the error
|
||||
// that prevented hashing it) from the hash workers to the hash phase.
|
||||
// hashResult carries the hashes computed for one inode run (or the
|
||||
// error that prevented computing them) from the pool's workers to the
|
||||
// phase that started the pool: head, tail, and content from
|
||||
// hashSignature, content alone from hashContentOnly.
|
||||
type hashResult struct {
|
||||
run []fileRec
|
||||
head string
|
||||
tail string
|
||||
content string
|
||||
err error
|
||||
}
|
||||
|
||||
@@ -856,10 +1211,10 @@ type hashPool struct {
|
||||
}
|
||||
|
||||
// startHashPool starts the feeder and the workers over runs. Workers
|
||||
// hash each run's first path (all paths in a run are hard links to the
|
||||
// same inode) and write one result per run.
|
||||
func startHashPool(ctx context.Context, runs [][]fileRec,
|
||||
workers int,
|
||||
// hash each run's first path with hash (all paths in a run are hard
|
||||
// links to the same inode) and write one result per run.
|
||||
func startHashPool(ctx context.Context, runs [][]fileRec, workers int,
|
||||
hash func(path string, size int64) (string, string, string, error),
|
||||
) *hashPool {
|
||||
ctx, cancel := context.WithCancel(ctx)
|
||||
|
||||
@@ -871,7 +1226,7 @@ func startHashPool(ctx context.Context, runs [][]fileRec,
|
||||
wg.Go(func() { feedHashJobs(ctx, runs, jobs) })
|
||||
|
||||
for range workers {
|
||||
wg.Go(func() { hashWorker(ctx, jobs, results) })
|
||||
wg.Go(func() { hashWorker(ctx, jobs, results, hash) })
|
||||
}
|
||||
|
||||
done := make(chan struct{})
|
||||
@@ -917,24 +1272,25 @@ func feedHashJobs(ctx context.Context, runs [][]fileRec,
|
||||
}
|
||||
}
|
||||
|
||||
// hashWorker hashes one inode run at a time until jobs is closed or the
|
||||
// scan is cancelled. A cancelled worker drops the runs still queued
|
||||
// instead of stopping its reads of jobs: the range must run out for the
|
||||
// pool to tear down, and reading a file nobody wants the hash of only
|
||||
// delays that.
|
||||
// hashWorker hashes one inode run at a time with hash until jobs is
|
||||
// closed or the scan is cancelled. A cancelled worker drops the runs
|
||||
// still queued instead of stopping its reads of jobs: the range must
|
||||
// run out for the pool to tear down, and reading a file nobody wants
|
||||
// the hash of only delays that.
|
||||
func hashWorker(ctx context.Context, jobs <-chan []fileRec,
|
||||
results chan<- hashResult,
|
||||
hash func(path string, size int64) (string, string, string, error),
|
||||
) {
|
||||
for run := range jobs {
|
||||
if ctx.Err() != nil {
|
||||
continue
|
||||
}
|
||||
|
||||
head, tail, err := hashHeadTail(run[0].path, run[0].size)
|
||||
head, tail, content, err := hash(run[0].path, run[0].size)
|
||||
|
||||
select {
|
||||
case results <- hashResult{
|
||||
run: run, head: head, tail: tail, err: err,
|
||||
run: run, head: head, tail: tail, content: content, err: err,
|
||||
}:
|
||||
case <-ctx.Done():
|
||||
return
|
||||
@@ -942,55 +1298,151 @@ func hashWorker(ctx context.Context, jobs <-chan []fileRec,
|
||||
}
|
||||
}
|
||||
|
||||
// emptyHash is the lowercase-hex SHA-256 of the empty input: the head
|
||||
// and tail hash of every zero-length file.
|
||||
// emptyHash is the lowercase-hex SHA-256 of the empty input: the head,
|
||||
// tail, and content hash of every zero-length file.
|
||||
const emptyHash = "e3b0c44298fc1c149afbf4c8996fb924" +
|
||||
"27ae41e4649b934ca495991b7852b855"
|
||||
|
||||
// hashHeadTail returns the lowercase-hex SHA-256 of the first
|
||||
// min(chunk, size) bytes and of the last min(chunk, size) bytes of the
|
||||
// file at path. The two reads overlap when size < 2*chunk. size is the
|
||||
// value recorded when the file was statted; a zero-length file's
|
||||
// hashes are constant, so it is never even opened.
|
||||
func hashHeadTail(path string, size int64) (string, string, error) {
|
||||
// hashSignature computes the hashes the hash phase records for a file
|
||||
// whose size is shared; with the file size they form its duplicate
|
||||
// signature. A file below headTailMin is hashed in full and its
|
||||
// whole-file SHA-256 is returned as head, tail, and content alike —
|
||||
// that range takes no separate end-window step. For a file at or above
|
||||
// headTailMin only the head and tail are computed, the SHA-256 of its
|
||||
// first and last headTailWindow bytes, and content is returned empty:
|
||||
// the content phase computes it with hashContentOnly once the file's
|
||||
// size, head, and tail match another file's. Two files are duplicates
|
||||
// only when all four agree; any mismatch means not a duplicate. size
|
||||
// is the value recorded when the file was statted; a zero-length file
|
||||
// has constant hashes and is never opened.
|
||||
func hashSignature(path string, size int64) (string, string, string, error) {
|
||||
if size == 0 {
|
||||
return emptyHash, emptyHash, nil
|
||||
return emptyHash, emptyHash, emptyHash, nil
|
||||
}
|
||||
|
||||
//nolint:gosec // hashing operator-supplied paths is the tool's purpose
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
return "", "", "", err
|
||||
}
|
||||
|
||||
defer func() { _ = f.Close() }()
|
||||
|
||||
n := min(int64(chunk), size)
|
||||
// Below the threshold the whole file is hashed directly, with no
|
||||
// end-window step: head and tail both carry the whole-file hash.
|
||||
if size < int64(headTailMin) {
|
||||
content, err := hashWhole(f, size)
|
||||
if err != nil {
|
||||
return "", "", "", err
|
||||
}
|
||||
|
||||
buf := make([]byte, n)
|
||||
return content, content, content, nil
|
||||
}
|
||||
|
||||
_, err = f.ReadAt(buf, 0)
|
||||
head, tail, err := hashEnds(f, size)
|
||||
if err != nil {
|
||||
return "", "", "", err
|
||||
}
|
||||
|
||||
return head, tail, "", nil
|
||||
}
|
||||
|
||||
// hashContentOnly returns the content hash of the file at path, which
|
||||
// is at least headTailMin bytes: the content phase's read. head and
|
||||
// tail are returned empty, because the content phase keeps the ones its
|
||||
// records already hold.
|
||||
func hashContentOnly(path string, size int64) (string, string, string, error) {
|
||||
//nolint:gosec // hashing operator-supplied paths is the tool's purpose
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return "", "", "", err
|
||||
}
|
||||
|
||||
defer func() { _ = f.Close() }()
|
||||
|
||||
content, err := hashContent(f, size)
|
||||
|
||||
return "", "", content, err
|
||||
}
|
||||
|
||||
// hashEnds returns the SHA-256 of the first and last headTailWindow
|
||||
// bytes of f. It is called only for files at least headTailMin, which
|
||||
// is far larger than two windows, so the windows never overlap and both
|
||||
// reads are always full.
|
||||
func hashEnds(f *os.File, size int64) (string, string, error) {
|
||||
buf := make([]byte, headTailWindow)
|
||||
|
||||
_, err := f.ReadAt(buf, 0)
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
}
|
||||
|
||||
h := sha256.Sum256(buf)
|
||||
head := hex.EncodeToString(h[:])
|
||||
|
||||
// When the whole file fits in one chunk the tail window is exactly
|
||||
// the bytes just read: reuse the head hash instead of issuing a
|
||||
// second read for every small file.
|
||||
if size <= int64(chunk) {
|
||||
hh := hex.EncodeToString(h[:])
|
||||
|
||||
return hh, hh, nil
|
||||
}
|
||||
|
||||
_, err = f.ReadAt(buf, size-n)
|
||||
_, err = f.ReadAt(buf, size-int64(headTailWindow))
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
}
|
||||
|
||||
t := sha256.Sum256(buf)
|
||||
|
||||
return hex.EncodeToString(h[:]), hex.EncodeToString(t[:]), nil
|
||||
return head, hex.EncodeToString(t[:]), nil
|
||||
}
|
||||
|
||||
// hashContent returns the content-rung hash of f: the SHA-256 of the
|
||||
// whole file when it is smaller than wholeFileMax, or of sampled
|
||||
// windows when it is that size or larger.
|
||||
func hashContent(f *os.File, size int64) (string, error) {
|
||||
if size >= int64(wholeFileMax) {
|
||||
return hashSamples(f, size)
|
||||
}
|
||||
|
||||
return hashWhole(f, size)
|
||||
}
|
||||
|
||||
// hashWhole returns the SHA-256 of the entire file. A SectionReader is
|
||||
// used so the read is independent of the offset left by any end-window
|
||||
// reads. Reading fewer than size bytes means the file shrank between
|
||||
// the stat and the hash; that is an error rather than a hash of content
|
||||
// that no longer matches the recorded size.
|
||||
func hashWhole(f *os.File, size int64) (string, error) {
|
||||
h := sha256.New()
|
||||
|
||||
n, err := io.Copy(h, io.NewSectionReader(f, 0, size))
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
if n != size {
|
||||
return "", fmt.Errorf("read %d of %d bytes: %w", n, size,
|
||||
io.ErrUnexpectedEOF)
|
||||
}
|
||||
|
||||
return hex.EncodeToString(h.Sum(nil)), nil
|
||||
}
|
||||
|
||||
// hashSamples feeds sampleWindow bytes at each gigabyte-aligned offset
|
||||
// (0, sampleStride, 2*sampleStride, ... while inside the file), in
|
||||
// order, into one hash, each window truncated at end of file. This is
|
||||
// the probabilistic large-file rung: two files of equal size agreeing
|
||||
// on every sample are reported as duplicates without every byte being
|
||||
// read. Because size is part of the signature, files of different sizes
|
||||
// never reach this comparison, so the sample boundaries always align.
|
||||
func hashSamples(f *os.File, size int64) (string, error) {
|
||||
h := sha256.New()
|
||||
buf := make([]byte, sampleWindow)
|
||||
|
||||
for off := int64(0); off < size; off += int64(sampleStride) {
|
||||
n := min(int64(sampleWindow), size-off)
|
||||
|
||||
_, err := f.ReadAt(buf[:n], off)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
h.Write(buf[:n])
|
||||
}
|
||||
|
||||
return hex.EncodeToString(h.Sum(nil)), nil
|
||||
}
|
||||
|
||||
+833
-69
File diff suppressed because it is too large
Load Diff
+13
-42
@@ -3,20 +3,16 @@
|
||||
# this repo. Idempotent: every install is guarded by a check so already
|
||||
# installed tools are skipped. Base tooling comes from nix, apt, brew,
|
||||
# or apk (detected in that order); assumes nothing is present (not git,
|
||||
# make, or go). The linter is NOT installed: golangci-lint runs via
|
||||
# docker only (script/lint), pinned by image digest, so the only lint
|
||||
# make, or go). Neither the linter nor the Markdown formatter is
|
||||
# installed: golangci-lint (script/lint) and prettier (script/fmt,
|
||||
# script/fmt-check) run via docker only, pinned by hash, so their only
|
||||
# prerequisite is a working docker — which is warned about, not
|
||||
# installed, because everything except linting works without it.
|
||||
# installed, because everything except linting and formatting works
|
||||
# without it.
|
||||
set -eu
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
|
||||
# yarn provides prettier, which formats Markdown. yarn is a tool, like
|
||||
# node/git/make/go below; the reference that governs formatting output is
|
||||
# prettier, pinned by yarn.lock's integrity hash and installed by
|
||||
# `yarn install --frozen-lockfile`.
|
||||
YARN_VERSION="1.22.22"
|
||||
|
||||
PKGMGR=""
|
||||
SUDO=""
|
||||
APT_UPDATED=""
|
||||
@@ -64,21 +60,6 @@ missing() {
|
||||
! command -v "$1" >/dev/null 2>&1
|
||||
}
|
||||
|
||||
ensure_node() {
|
||||
if ! missing node; then return 0; fi
|
||||
pkg_install nodejs nodejs node nodejs
|
||||
}
|
||||
|
||||
ensure_yarn() {
|
||||
if ! missing yarn; then return 0; fi
|
||||
if ! missing corepack; then
|
||||
corepack enable >/dev/null 2>&1 || true
|
||||
corepack prepare "yarn@$YARN_VERSION" --activate
|
||||
else
|
||||
pkg_install yarn yarn yarn yarn
|
||||
fi
|
||||
}
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
|
||||
@@ -92,25 +73,15 @@ main() {
|
||||
if missing make; then pkg_install gnumake make make make; fi
|
||||
if missing go; then pkg_install go golang go go; fi
|
||||
|
||||
# node runs prettier and is an unpinned host tool for the same reason
|
||||
# git/make/go are: it comes from the host package manager, whatever
|
||||
# version it ships. It is not installed via nvm the way the canonical
|
||||
# template does, because nvm's prebuilt node is glibc-linked and does
|
||||
# not run on this repo's musl/Alpine build image. prettier — the tool
|
||||
# whose version affects formatting output — is pinned by yarn.lock.
|
||||
ensure_node
|
||||
ensure_yarn
|
||||
yarn install --frozen-lockfile
|
||||
|
||||
# Linting runs via docker only (script/lint), so docker is a lint
|
||||
# prerequisite rather than something bootstrap installs. Warn, do
|
||||
# not fail: everything except `make lint` — and, through it,
|
||||
# `make check`, `make docker` and the pre-commit hook — works
|
||||
# without it.
|
||||
# Linting and Markdown formatting run via docker only, so docker is
|
||||
# their prerequisite rather than something bootstrap installs. Warn,
|
||||
# do not fail: everything except `make lint`, `make fmt` and
|
||||
# `make fmt-check` — and, through them, `make check`, `make docker`
|
||||
# and the pre-commit hook — works without it.
|
||||
if missing docker; then
|
||||
echo "bootstrap: WARNING: docker not found; make lint, make check" >&2
|
||||
echo "bootstrap: and make docker require it. Install docker to" >&2
|
||||
echo "bootstrap: run the linter." >&2
|
||||
echo "bootstrap: WARNING: docker not found; make lint, make fmt," >&2
|
||||
echo "bootstrap: make fmt-check, make check and make docker" >&2
|
||||
echo "bootstrap: require it." >&2
|
||||
fi
|
||||
|
||||
go mod download
|
||||
|
||||
+14
-14
@@ -3,23 +3,23 @@
|
||||
# push.
|
||||
#
|
||||
# The Dockerfile runs the gates individually as build steps, not the
|
||||
# make check aggregate: the lint stage runs make fmt-check,
|
||||
# make check aggregate: the lint stage runs the gofmt check,
|
||||
# script/verify-lint-image-pin, golangci-lint config verify and
|
||||
# golangci-lint run; the build stage, dropped to an unprivileged user,
|
||||
# runs make test and make fmt-check. Neither make lint nor make check
|
||||
# appears, because both reach script/lint, which is itself a docker
|
||||
# build, and a docker build cannot run inside one. Lint is not skipped
|
||||
# by that — the linter is invoked directly in the lint stage, and the
|
||||
# build stage's COPY --from=lint makes that stage a prerequisite, so
|
||||
# BuildKit must finish it first. Between the two stages everything
|
||||
# make check would run has run, which is why a successful build here
|
||||
# implies the repo is green.
|
||||
# golangci-lint run; the markdown stage runs the prettier check; the
|
||||
# build stage, dropped to an unprivileged user, runs make test. None of
|
||||
# make lint, make fmt-check or make check appears, because each runs
|
||||
# docker, and docker cannot run inside a docker build. Nothing is
|
||||
# skipped by that — the linter, gofmt and prettier are invoked directly
|
||||
# in their stages, and the build stage's COPY --from lines make those
|
||||
# stages prerequisites, so BuildKit must finish them first. Between the
|
||||
# three stages everything make check would run has run, which is why a
|
||||
# successful build here implies the repo is green.
|
||||
#
|
||||
# That implication holds only because of CHECK_EPOCH. A COPY layer is
|
||||
# invalidated by changed content, and a merge commit's tree is
|
||||
# byte-identical to the branch head it merges, so without a fresh value
|
||||
# here Docker serves the gate layers from cache and the build reports a
|
||||
# green it never earned. Passing the current epoch invalidates the gate
|
||||
# invalidated only by changed content, and a rebuild of an unchanged
|
||||
# checkout sends the same content, so without a fresh value here Docker
|
||||
# serves the gate layers from cache and the build reports a green it
|
||||
# never earned. Passing the current epoch invalidates the gate
|
||||
# layers on every run while leaving the pinned base images and
|
||||
# go mod download cached; see the Dockerfile for the placement.
|
||||
set -eu
|
||||
|
||||
+5
-5
@@ -4,11 +4,11 @@
|
||||
#
|
||||
# CHECK_EPOCH is passed for the same reason script/cibuild passes it:
|
||||
# without it Docker serves the Dockerfile's gate layers from cache on an
|
||||
# unchanged tree and this exits 0 having run neither the lint stage's
|
||||
# gates nor the builder stage's test and fmt-check gates. This is the
|
||||
# set of gates a developer or reviewer runs by hand, so a cached pass
|
||||
# here is the most misleading result the repo can produce. Dependency
|
||||
# layers sit above the ARG and stay cached.
|
||||
# unchanged tree and this exits 0 having run none of the lint stage's
|
||||
# gates, the markdown stage's prettier gate or the builder stage's test
|
||||
# gate. This is the set of gates a developer or reviewer runs by hand,
|
||||
# so a cached pass here is the most misleading result the repo can
|
||||
# produce. Dependency layers sit above the ARG and stay cached.
|
||||
set -eu
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||
|
||||
+12
-13
@@ -1,23 +1,22 @@
|
||||
#!/bin/sh
|
||||
# script/fmt: format all files (writes). gofmt for Go, prettier for
|
||||
# Markdown. prettier is the pinned devDependency in package.json/
|
||||
# yarn.lock; script/bootstrap installs it (see run_prettier).
|
||||
# script/fmt: format all files (writes): the Go sources with gofmt, the
|
||||
# Markdown with prettier. prettier is never installed on the host: it
|
||||
# runs from the Dockerfile's prettier stage with the repository mounted,
|
||||
# as the calling user so the files it rewrites keep their owner. The tag
|
||||
# makes each build replace the previous image instead of leaving another
|
||||
# one behind.
|
||||
set -eu
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
|
||||
run_prettier() {
|
||||
if ! command -v yarn >/dev/null 2>&1; then
|
||||
echo "fmt: yarn not found; run script/bootstrap first" >&2
|
||||
exit 1
|
||||
fi
|
||||
yarn run prettier "$@"
|
||||
}
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
gofmt -s -w .
|
||||
run_prettier --write '**/*.md' --tab-width 4 --prose-wrap always
|
||||
image="$("$SCRIPT_DIR/projectname")-prettier"
|
||||
docker build -q --target prettier -t "$image" . >/dev/null
|
||||
docker run --rm --user "$(id -u):$(id -g)" -v "$ROOT:/src" "$image" \
|
||||
prettier --write '**/*.md' --tab-width 4 --prose-wrap always
|
||||
}
|
||||
|
||||
main "$@"
|
||||
|
||||
+14
-17
@@ -1,37 +1,34 @@
|
||||
#!/bin/sh
|
||||
# script/fmt-check: check formatting (read-only). Same scope as
|
||||
# script/fmt: gofmt for Go, prettier for Markdown. Both run every time
|
||||
# and each reports independently, so a failure names which formatter is
|
||||
# unhappy; the script exits non-zero if either found unformatted files.
|
||||
# script/fmt, but fails instead of writing. gofmt and prettier both run
|
||||
# every time and each reports its own failure, so the output says which
|
||||
# one failed.
|
||||
set -eu
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
|
||||
run_prettier() {
|
||||
if ! command -v yarn >/dev/null 2>&1; then
|
||||
echo "fmt-check: yarn not found; run script/bootstrap first" >&2
|
||||
exit 1
|
||||
fi
|
||||
yarn run prettier "$@"
|
||||
}
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
rc=0
|
||||
status=0
|
||||
|
||||
files="$(gofmt -s -l .)"
|
||||
if [ -n "$files" ]; then
|
||||
echo "gofmt: files not formatted:" >&2
|
||||
echo "$files" >&2
|
||||
rc=1
|
||||
status=1
|
||||
fi
|
||||
|
||||
if ! run_prettier --check '**/*.md' --tab-width 4 --prose-wrap always; then
|
||||
# Same image as script/fmt; see there.
|
||||
image="$("$SCRIPT_DIR/projectname")-prettier"
|
||||
docker build -q --target prettier -t "$image" . >/dev/null
|
||||
if ! docker run --rm -v "$ROOT:/src:ro" "$image" \
|
||||
prettier --check '**/*.md' --tab-width 4 --prose-wrap always; then
|
||||
echo "prettier: Markdown not formatted; run make fmt" >&2
|
||||
rc=1
|
||||
status=1
|
||||
fi
|
||||
|
||||
exit "$rc"
|
||||
exit "$status"
|
||||
}
|
||||
|
||||
main "$@"
|
||||
|
||||
@@ -5,48 +5,59 @@ import (
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"slices"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// fileSig is a file's duplicate signature; mtime is excluded.
|
||||
type fileSig struct {
|
||||
size int64
|
||||
head string
|
||||
tail string
|
||||
}
|
||||
|
||||
// treeNode is one directory reconstructed from the scan stream.
|
||||
// treeNode is one directory reconstructed from the record paths.
|
||||
type treeNode struct {
|
||||
path string
|
||||
parent *treeNode
|
||||
dirs map[string]*treeNode
|
||||
files map[string]fileSig
|
||||
// entries holds the serialized child entries until the digest is
|
||||
// computed from them, and is then dropped.
|
||||
entries []string
|
||||
digest [sha256.Size]byte
|
||||
fileCount int64
|
||||
totalSize int64
|
||||
}
|
||||
|
||||
// runTrees implements the trees subcommand: it reads every record from
|
||||
// the database, reconstructs the directory hierarchy from the record
|
||||
// paths, computes a Merkle-style digest per directory, and prints
|
||||
// maximal duplicate-tree groups as TSV on stdout. It never touches the
|
||||
// scanned filesystem; its only I/O is the database, stdout, and
|
||||
// stderr.
|
||||
func runTrees(ctx context.Context) error {
|
||||
recs, err := loadRecords(ctx)
|
||||
// the database in path order, reconstructs the directory hierarchy from
|
||||
// the record paths, computes a Merkle-style digest per directory, and
|
||||
// prints maximal duplicate-tree groups as TSV on stdout. It never
|
||||
// touches the scanned filesystem; its only I/O is the database, stdout,
|
||||
// and stderr. Any database problem, including a missing database, is
|
||||
// fatal.
|
||||
func runTrees(ctx context.Context, stdout io.Writer) error {
|
||||
dbPath := databasePath()
|
||||
|
||||
db, err := openReportDatabase(ctx, dbPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
super, allDirs := buildHierarchy(recs)
|
||||
super.compute()
|
||||
defer func() { _ = db.Close() }()
|
||||
|
||||
records := 0
|
||||
tree := newTreeBuilder()
|
||||
|
||||
err = loadFileRows(ctx, db, func(r scanRec) {
|
||||
records++
|
||||
|
||||
tree.add(r)
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("database %s: %w", dbPath, err)
|
||||
}
|
||||
|
||||
super, allDirs := tree.finish()
|
||||
|
||||
dupes := collectTreeGroups(allDirs, super)
|
||||
|
||||
out := bufio.NewWriterSize(os.Stdout, ioBufSize)
|
||||
out := bufio.NewWriterSize(stdout, ioBufSize)
|
||||
|
||||
_, err = fmt.Fprintln(out, "first\tdupe\tfiles\tsize")
|
||||
if err != nil {
|
||||
@@ -61,7 +72,8 @@ func runTrees(ctx context.Context) error {
|
||||
first := g[0]
|
||||
for _, n := range g[1:] {
|
||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\t%d\n",
|
||||
first.path, n.path, first.fileCount, first.totalSize)
|
||||
escapePath(first.path), escapePath(n.path),
|
||||
first.fileCount, first.totalSize)
|
||||
if err != nil {
|
||||
return fmt.Errorf("write stdout: %w", err)
|
||||
}
|
||||
@@ -79,63 +91,128 @@ func runTrees(ctx context.Context) error {
|
||||
fmt.Fprintf(os.Stderr,
|
||||
"trees: %d records read, %d duplicate tree groups, %d dupe trees, "+
|
||||
"%s reclaimable\n",
|
||||
len(recs), len(dupes), dupeTrees, humanBytes(reclaimable))
|
||||
records, len(dupes), dupeTrees, humanBytes(reclaimable))
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// buildHierarchy reconstructs the directory hierarchy from the record
|
||||
// paths under a synthetic super-root. Paths are split on "/"; for
|
||||
// absolute paths the first component is empty, which simply becomes a
|
||||
// top-level node representing "/". It returns the super-root and every
|
||||
// directory node created.
|
||||
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
||||
// treeBuilder reconstructs the directory hierarchy from records added
|
||||
// in path order, under a synthetic super-root. Paths are split on "/";
|
||||
// for absolute paths the first component is empty, which becomes the
|
||||
// top-level directory with path "/". In path order all the paths under
|
||||
// one directory come together, so a directory is complete once a path
|
||||
// outside it is added: its digest is computed then and its entries are
|
||||
// dropped. Only the directories holding the latest path keep entries.
|
||||
type treeBuilder struct {
|
||||
super *treeNode
|
||||
// open lists the directories holding the latest path, outermost
|
||||
// first, starting with the super-root; names[i] is open[i]'s name.
|
||||
open []*treeNode
|
||||
names []string
|
||||
// dirs lists every completed directory.
|
||||
dirs []*treeNode
|
||||
}
|
||||
|
||||
func newTreeBuilder() *treeBuilder {
|
||||
super := &treeNode{}
|
||||
|
||||
var allDirs []*treeNode
|
||||
return &treeBuilder{
|
||||
super: super,
|
||||
open: []*treeNode{super},
|
||||
names: []string{""},
|
||||
}
|
||||
}
|
||||
|
||||
for _, r := range recs {
|
||||
// add adds one record. Each record must come after the previous one in
|
||||
// path order (byte order); otherwise a completed directory would be
|
||||
// started again as a second directory with the same path.
|
||||
func (b *treeBuilder) add(r scanRec) {
|
||||
comps := strings.Split(r.path, "/")
|
||||
dirNames, name := comps[:len(comps)-1], comps[len(comps)-1]
|
||||
|
||||
node := super
|
||||
for _, c := range comps[:len(comps)-1] {
|
||||
child := node.dirs[c]
|
||||
if child == nil {
|
||||
childPath := c
|
||||
if node != super {
|
||||
childPath = node.path + "/" + c
|
||||
// Keep the open directories that hold this path; complete the rest.
|
||||
depth := 1
|
||||
for depth < len(b.open) && depth <= len(dirNames) &&
|
||||
b.names[depth] == dirNames[depth-1] {
|
||||
depth++
|
||||
}
|
||||
|
||||
child = &treeNode{path: childPath, parent: node}
|
||||
if node.dirs == nil {
|
||||
node.dirs = make(map[string]*treeNode)
|
||||
b.closeTo(depth)
|
||||
|
||||
for _, c := range dirNames[depth-1:] {
|
||||
b.openDir(c)
|
||||
}
|
||||
|
||||
node.dirs[c] = child
|
||||
allDirs = append(allDirs, child)
|
||||
dir := b.open[len(b.open)-1]
|
||||
dir.entries = append(dir.entries, fileEntry(name, r))
|
||||
dir.fileCount++
|
||||
dir.totalSize += r.size
|
||||
}
|
||||
|
||||
node = child
|
||||
// openDir starts the directory called name inside the innermost open
|
||||
// one.
|
||||
func (b *treeBuilder) openDir(name string) {
|
||||
parent := b.open[len(b.open)-1]
|
||||
path := parent.path + "/" + name
|
||||
|
||||
// The root directory's path is "/", not empty, and its children's
|
||||
// paths start with one slash, not two.
|
||||
switch {
|
||||
case parent == b.super && name == "":
|
||||
path = "/"
|
||||
case parent == b.super:
|
||||
path = name
|
||||
case parent.path == "/":
|
||||
path = "/" + name
|
||||
}
|
||||
|
||||
if node.files == nil {
|
||||
node.files = make(map[string]fileSig)
|
||||
b.open = append(b.open, &treeNode{path: path, parent: parent})
|
||||
b.names = append(b.names, name)
|
||||
}
|
||||
|
||||
sig := fileSig{size: r.size, head: r.head, tail: r.tail}
|
||||
// closeTo completes the open directories after the first n, innermost
|
||||
// first: each one's digest is computed and entered in its parent along
|
||||
// with its totals.
|
||||
func (b *treeBuilder) closeTo(n int) {
|
||||
for len(b.open) > n {
|
||||
last := len(b.open) - 1
|
||||
dir, name := b.open[last], b.names[last]
|
||||
b.open, b.names = b.open[:last], b.names[:last]
|
||||
|
||||
// An unhashed record (its size was unique when last scanned)
|
||||
// has unknown content: give it a signature no other file can
|
||||
// share, so trees containing it never compare equal. Real
|
||||
// heads are hex, so the NUL-prefixed form cannot collide.
|
||||
if sig.head == "" {
|
||||
sig.head = "unhashed\x00" + r.path
|
||||
dir.computeDigest()
|
||||
|
||||
dir.parent.entries = append(dir.parent.entries,
|
||||
"d\x00"+name+"\x00"+string(dir.digest[:]))
|
||||
dir.parent.fileCount += dir.fileCount
|
||||
dir.parent.totalSize += dir.totalSize
|
||||
b.dirs = append(b.dirs, dir)
|
||||
}
|
||||
}
|
||||
|
||||
node.files[comps[len(comps)-1]] = sig
|
||||
// finish completes every open directory and returns the super-root and
|
||||
// every directory.
|
||||
func (b *treeBuilder) finish() (*treeNode, []*treeNode) {
|
||||
b.closeTo(1)
|
||||
|
||||
return b.super, b.dirs
|
||||
}
|
||||
|
||||
return super, allDirs
|
||||
// fileEntry serializes a file child for its directory's digest: its
|
||||
// name and its signature (size, head, tail, content); mtime is
|
||||
// excluded.
|
||||
func fileEntry(name string, r scanRec) string {
|
||||
content := r.content
|
||||
|
||||
// A record without a content hash has unknown content (README
|
||||
// "Database"): give it a signature no other file can share, so
|
||||
// trees containing it never compare equal. Real hashes are hex, so
|
||||
// the NUL-prefixed form cannot collide.
|
||||
if content == "" {
|
||||
content = "unhashed\x00" + r.path
|
||||
}
|
||||
|
||||
return "f\x00" + name + "\x00" + strconv.FormatInt(r.size, 10) +
|
||||
"\x00" + r.head + "\x00" + r.tail + "\x00" + content
|
||||
}
|
||||
|
||||
// collectTreeGroups groups directories by digest and returns every
|
||||
@@ -177,38 +254,22 @@ func collectTreeGroups(allDirs []*treeNode, super *treeNode) [][]*treeNode {
|
||||
return dupes
|
||||
}
|
||||
|
||||
// compute fills in digest, fileCount, and totalSize for n and all of
|
||||
// its descendants. A directory's digest is the SHA-256 of its child
|
||||
// entries — files serialized with name and signature, subdirectories
|
||||
// with name and recursive digest — sorted byte-lexicographically.
|
||||
// Filenames cannot contain NUL or "/", so NUL delimiters are
|
||||
// unambiguous.
|
||||
func (n *treeNode) compute() {
|
||||
entries := make([]string, 0, len(n.dirs)+len(n.files))
|
||||
for name, sig := range n.files {
|
||||
entries = append(entries,
|
||||
"f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+
|
||||
"\x00"+sig.head+"\x00"+sig.tail)
|
||||
n.fileCount++
|
||||
n.totalSize += sig.size
|
||||
}
|
||||
|
||||
for name, child := range n.dirs {
|
||||
child.compute()
|
||||
entries = append(entries, "d\x00"+name+"\x00"+string(child.digest[:]))
|
||||
n.fileCount += child.fileCount
|
||||
n.totalSize += child.totalSize
|
||||
}
|
||||
|
||||
slices.Sort(entries)
|
||||
// computeDigest sets n's digest and drops its entries. A directory's
|
||||
// digest is the SHA-256 of its child entries — files serialized with
|
||||
// name and signature, subdirectories with name and recursive digest —
|
||||
// sorted byte-lexicographically. Filenames cannot contain NUL or "/",
|
||||
// so NUL delimiters are unambiguous.
|
||||
func (n *treeNode) computeDigest() {
|
||||
slices.Sort(n.entries)
|
||||
|
||||
h := sha256.New()
|
||||
for _, e := range entries {
|
||||
for _, e := range n.entries {
|
||||
h.Write([]byte(e))
|
||||
h.Write([]byte{0})
|
||||
}
|
||||
|
||||
copy(n.digest[:], h.Sum(nil))
|
||||
n.entries = nil
|
||||
}
|
||||
|
||||
// suppressed reports whether a duplicate-tree group is non-maximal: its
|
||||
|
||||
+145
-30
@@ -1,6 +1,8 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"database/sql"
|
||||
"slices"
|
||||
"testing"
|
||||
)
|
||||
@@ -9,23 +11,56 @@ import (
|
||||
const (
|
||||
f1Head = "f1h"
|
||||
f1Tail = "f1t"
|
||||
f1Content = "f1c"
|
||||
f2Head = "f2h"
|
||||
f2Tail = "f2t"
|
||||
f2Content = "f2c"
|
||||
)
|
||||
|
||||
// smokeTreeRecs mirrors the README smoke-test tree layout: /d/t1 and
|
||||
// /d/t2 are identical, /d/t3 differs from them only by one filename.
|
||||
func smokeTreeRecs() []scanRec {
|
||||
return []scanRec{
|
||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t1/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t1/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t2/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t2/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t3/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t3/sub/f2renamed"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t1/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t1/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t2/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t2/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t3/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, content: f2Content,
|
||||
path: "/d/t3/sub/f2renamed"},
|
||||
}
|
||||
}
|
||||
|
||||
// dbTree builds the directory hierarchy from the records in db the way
|
||||
// trees does, and returns the super-root and every directory.
|
||||
func dbTree(t *testing.T, db *sql.DB) (*treeNode, []*treeNode) {
|
||||
t.Helper()
|
||||
|
||||
tree := newTreeBuilder()
|
||||
|
||||
err := loadFileRows(t.Context(), db, tree.add)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return tree.finish()
|
||||
}
|
||||
|
||||
// treeOf writes recs into a fresh database and builds the directory
|
||||
// hierarchy from it the way trees does.
|
||||
func treeOf(t *testing.T, recs []scanRec) (*treeNode, []*treeNode) {
|
||||
t.Helper()
|
||||
|
||||
db := openTestDB(t)
|
||||
|
||||
err := applyChanges(t.Context(), db, recs, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return dbTree(t, db)
|
||||
}
|
||||
|
||||
// nodeByPath finds the directory node with the given path.
|
||||
func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode {
|
||||
t.Helper()
|
||||
@@ -56,11 +91,10 @@ func groupPaths(groups [][]*treeNode) [][]string {
|
||||
return out
|
||||
}
|
||||
|
||||
func TestBuildHierarchyCounts(t *testing.T) {
|
||||
func TestTreeCounts(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
super, dirs := buildHierarchy(smokeTreeRecs())
|
||||
super.compute()
|
||||
_, dirs := treeOf(t, smokeTreeRecs())
|
||||
|
||||
d := nodeByPath(t, dirs, "/d")
|
||||
if d.fileCount != 6 || d.totalSize != 9300 {
|
||||
@@ -81,11 +115,98 @@ func TestBuildHierarchyCounts(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestTreeRootPath(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// The root directory's path is "/", never empty, and its
|
||||
// children's paths start with a single slash.
|
||||
_, dirs := treeOf(t, []scanRec{{path: "/f"}, {path: "/srv/g"}})
|
||||
|
||||
got := make([]string, 0, len(dirs))
|
||||
for _, d := range dirs {
|
||||
got = append(got, d.path)
|
||||
}
|
||||
|
||||
slices.Sort(got)
|
||||
|
||||
want := []string{"/", "/srv"}
|
||||
if !slices.Equal(got, want) {
|
||||
t.Fatalf("directory paths = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTreeNamesSortingBeforeSlash(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// In path order "/a/b-x/f" and "/a/b.txt" come between the file
|
||||
// "/a/b" and "/a/b/f", because "-" and "." sort before "/". Each
|
||||
// directory must still be built once, whole, so /a matches /c.
|
||||
recs := make([]scanRec, 0, 8)
|
||||
|
||||
for _, top := range []string{"/a", "/c"} {
|
||||
for _, p := range []string{"/b", "/b-x/f", "/b.txt", "/b/f"} {
|
||||
content := "c"
|
||||
if p == "/b-x/f" {
|
||||
content = "other"
|
||||
}
|
||||
|
||||
recs = append(recs, scanRec{
|
||||
size: 1, head: "h", tail: "t", content: content, path: top + p,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
super, dirs := treeOf(t, recs)
|
||||
|
||||
got := make([]string, 0, len(dirs))
|
||||
for _, d := range dirs {
|
||||
got = append(got, d.path)
|
||||
}
|
||||
|
||||
slices.Sort(got)
|
||||
|
||||
want := []string{"/", "/a", "/a/b", "/a/b-x", "/c", "/c/b", "/c/b-x"}
|
||||
if !slices.Equal(got, want) {
|
||||
t.Fatalf("directory paths = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
groups := collectTreeGroups(dirs, super)
|
||||
|
||||
gotGroups := groupPaths(groups)
|
||||
wantGroups := [][]string{{"/a", "/c"}}
|
||||
|
||||
if !slices.EqualFunc(gotGroups, wantGroups, slices.Equal) {
|
||||
t.Fatalf("groups = %v, want %v", gotGroups, wantGroups)
|
||||
}
|
||||
|
||||
if groups[0][0].fileCount != 4 || groups[0][0].totalSize != 4 {
|
||||
t.Errorf("group totals: %d files %d bytes, want 4 4",
|
||||
groups[0][0].fileCount, groups[0][0].totalSize)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunTreesEscapesPaths(t *testing.T) {
|
||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{cmdTrees}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
}
|
||||
|
||||
want := "first\tdupe\tfiles\tsize\n" +
|
||||
`/d/\tone\ntwo\rthree\\four` + "\t/d/A\t1\t5\n"
|
||||
if got := stdout.String(); got != want {
|
||||
t.Errorf("stdout = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTreeDigests(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
super, dirs := buildHierarchy(smokeTreeRecs())
|
||||
super.compute()
|
||||
_, dirs := treeOf(t, smokeTreeRecs())
|
||||
|
||||
t1 := nodeByPath(t, dirs, "/d/t1")
|
||||
t2 := nodeByPath(t, dirs, "/d/t2")
|
||||
@@ -114,12 +235,11 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
|
||||
const sharedTail = "same"
|
||||
|
||||
recs := []scanRec{
|
||||
{size: 10, head: sharedTail, tail: sharedTail, path: "/r/a/f"},
|
||||
{size: 10, head: "DIFF", tail: sharedTail, path: "/r/b/f"},
|
||||
{size: 10, head: sharedTail, tail: sharedTail, content: "c", path: "/r/a/f"},
|
||||
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
|
||||
}
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
super.compute()
|
||||
_, dirs := treeOf(t, recs)
|
||||
|
||||
a := nodeByPath(t, dirs, "/r/a")
|
||||
b := nodeByPath(t, dirs, "/r/b")
|
||||
@@ -132,8 +252,7 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
|
||||
func TestCollectTreeGroupsMaximal(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
super, dirs := buildHierarchy(smokeTreeRecs())
|
||||
super.compute()
|
||||
super, dirs := treeOf(t, smokeTreeRecs())
|
||||
|
||||
groups := collectTreeGroups(dirs, super)
|
||||
|
||||
@@ -157,16 +276,14 @@ func TestCollectTreeGroupsDeterministic(t *testing.T) {
|
||||
|
||||
recs := smokeTreeRecs()
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
super.compute()
|
||||
super, dirs := treeOf(t, recs)
|
||||
|
||||
forward := groupPaths(collectTreeGroups(dirs, super))
|
||||
|
||||
reversed := slices.Clone(recs)
|
||||
slices.Reverse(reversed)
|
||||
|
||||
superR, dirsR := buildHierarchy(reversed)
|
||||
superR.compute()
|
||||
superR, dirsR := treeOf(t, reversed)
|
||||
|
||||
backward := groupPaths(collectTreeGroups(dirsR, superR))
|
||||
if !slices.EqualFunc(forward, backward, slices.Equal) {
|
||||
@@ -181,12 +298,11 @@ func TestCollectTreeGroupsSiblings(t *testing.T) {
|
||||
// Identical sibling dirs share a parent, so their group cannot be
|
||||
// implied by a parent group and must be reported.
|
||||
recs := []scanRec{
|
||||
{size: 10, head: "h", tail: "t", path: "/p/x1/f"},
|
||||
{size: 10, head: "h", tail: "t", path: "/p/x2/f"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x1/f"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
|
||||
}
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
super.compute()
|
||||
super, dirs := treeOf(t, recs)
|
||||
|
||||
got := groupPaths(collectTreeGroups(dirs, super))
|
||||
|
||||
@@ -203,13 +319,12 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) {
|
||||
// extra file, so the parents' digests differ and the x group must
|
||||
// be reported.
|
||||
recs := []scanRec{
|
||||
{size: 10, head: "h", tail: "t", path: "/p/a/x/f"},
|
||||
{size: 99, head: "e", tail: "e", path: "/p/a/extra"},
|
||||
{size: 10, head: "h", tail: "t", path: "/q/b/x/f"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/a/x/f"},
|
||||
{size: 99, head: "e", tail: "e", content: "e", path: "/p/a/extra"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
|
||||
}
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
super.compute()
|
||||
super, dirs := treeOf(t, recs)
|
||||
|
||||
got := groupPaths(collectTreeGroups(dirs, super))
|
||||
|
||||
|
||||
Reference in New Issue
Block a user