Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1f64f4d6de | ||
|
|
c737490a53 | ||
|
|
09a39ddf37 | ||
|
|
29a65016d0 | ||
|
|
7ac4f6b723 |
+6
-3
@@ -1,9 +1,12 @@
|
||||
.git
|
||||
# .git is sent without its config. Without a VERSION build argument the
|
||||
# stage that compiles runs `git describe --tags --always` on .git, which
|
||||
# does not need .git/config; that file can hold a credential, such as a
|
||||
# password in a remote URL or the token the CI checkout step stores there.
|
||||
.git/config
|
||||
|
||||
.claude
|
||||
.DS_Store
|
||||
sfdupes
|
||||
files.dat
|
||||
node_modules
|
||||
*.log
|
||||
*.out
|
||||
*.test
|
||||
|
||||
@@ -6,4 +6,6 @@ jobs:
|
||||
steps:
|
||||
# actions/checkout v4.2.2, 2026-02-22
|
||||
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- run: script/cibuild
|
||||
|
||||
@@ -27,7 +27,6 @@ node_modules/
|
||||
*.log
|
||||
|
||||
# Local scan data
|
||||
files.dat
|
||||
*.sqlite
|
||||
*.sqlite-shm
|
||||
*.sqlite-wal
|
||||
|
||||
@@ -1,2 +0,0 @@
|
||||
node_modules/
|
||||
yarn.lock
|
||||
@@ -1,4 +0,0 @@
|
||||
{
|
||||
"tabWidth": 4,
|
||||
"proseWrap": "always"
|
||||
}
|
||||
+19
-12
@@ -27,12 +27,9 @@ ARG CHECK_EPOCH
|
||||
# target now runs `docker build -f Dockerfile.lint`, and a docker build
|
||||
# cannot run a docker build: routing the gate through make would mean
|
||||
# nesting docker inside this image. Same reason `make check` is gone
|
||||
# from the build stage below.
|
||||
#
|
||||
# `make fmt-check` is not run in this stage: it now also runs prettier
|
||||
# over Markdown, and this golangci-lint image has no node. The gate runs
|
||||
# in the build stage below, where script/bootstrap installs node and
|
||||
# prettier.
|
||||
# from the build stage below. `make fmt-check` stays as it is — it is a
|
||||
# gate, not the aggregate, and it shells out to nothing.
|
||||
RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
|
||||
|
||||
# The FROM above and the one in Dockerfile.lint pin the same linter
|
||||
# twice, and nothing else keeps them in sync; this fails the build when
|
||||
@@ -80,12 +77,10 @@ COPY --from=lint /src/go.sum /dev/null
|
||||
# rather than duplicating the installs inline. Only script/ and the
|
||||
# dependency manifests are copied first, nothing else, so this layer
|
||||
# stays cached until the scripts or the dependencies change — bootstrap
|
||||
# runs `go mod download` and `yarn install`, which is why there is no
|
||||
# separate invocation of either here. The JS manifests (package.json,
|
||||
# yarn.lock) are copied too so the yarn install layer caches alongside
|
||||
# the Go one.
|
||||
# ends in `go mod download`, which is why there is no separate
|
||||
# invocation of it here.
|
||||
COPY script/ script/
|
||||
COPY go.mod go.sum package.json yarn.lock ./
|
||||
COPY go.mod go.sum ./
|
||||
RUN script/bootstrap
|
||||
|
||||
COPY . .
|
||||
@@ -113,7 +108,19 @@ ARG CHECK_EPOCH
|
||||
RUN echo "gate test, epoch ${CHECK_EPOCH}" && make test
|
||||
RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
|
||||
|
||||
RUN make build
|
||||
# The version stamped into the binary: the VERSION build argument when
|
||||
# one is given, otherwise `git describe --tags --always` of the .git in
|
||||
# the build context (git is installed by script/bootstrap above). A
|
||||
# context that carries .git and still yields no version fails the build;
|
||||
# with neither, as from a source tarball, it is "dev".
|
||||
ARG VERSION
|
||||
RUN version="${VERSION:-$(git describe --tags --always || echo dev)}"; \
|
||||
if [ -e .git ] && { [ -z "$version" ] || [ "$version" = dev ] || \
|
||||
[ "$version" = unknown ]; }; then \
|
||||
echo "no version could be derived although the build context carries .git" >&2; \
|
||||
exit 1; \
|
||||
fi; \
|
||||
make build VERSION="$version"
|
||||
|
||||
# Runtime stage
|
||||
# alpine:3.22, 2026-07-23
|
||||
|
||||
@@ -46,4 +46,4 @@ hooks:
|
||||
@script/install-precommit
|
||||
|
||||
clean:
|
||||
rm -f $(BINARY) files.dat
|
||||
rm -f $(BINARY)
|
||||
|
||||
@@ -1,386 +1,464 @@
|
||||
# Workflow
|
||||
|
||||
- take an issue from the `1.0.0` milestone on the tracker; work not yet on the
|
||||
tracker gets filed as an issue first
|
||||
- take an issue from the `1.0.0` milestone on the tracker; work not
|
||||
yet on the tracker gets filed as an issue first
|
||||
- branch (from `main`)
|
||||
- do the work, with tests, in small focused commits
|
||||
- record it at the top of Completed Steps (`TODO.md` changes in the same commit
|
||||
as the work)
|
||||
- push the branch and open a PR whose title ends with ` (closes #N)`
|
||||
- an independent review gates the merge; every finding is addressed or
|
||||
explicitly rebutted on the PR
|
||||
- record it at the top of Completed Steps (`TODO.md` changes in the
|
||||
same commit as the work)
|
||||
- push the branch and open a PR whose title ends with
|
||||
` (closes #N)`
|
||||
- an independent review gates the merge; every finding is addressed
|
||||
or explicitly rebutted on the PR
|
||||
- merge to `main` once the review passes
|
||||
|
||||
# Status
|
||||
|
||||
- pre-1.0
|
||||
- the Gitea tracker is authoritative for the pre-1.0 backlog: the open issues
|
||||
under the `1.0.0` milestone are what remains before the tag, and this file
|
||||
records history and process, not the queue
|
||||
- the Gitea tracker is authoritative for the pre-1.0 backlog: the
|
||||
open issues under the `1.0.0` milestone are what remains before
|
||||
the tag, and this file records history and process, not the queue
|
||||
|
||||
# Next Step
|
||||
|
||||
- take the next issue from the `1.0.0` milestone on the tracker:
|
||||
https://git.eeqj.de/sneak/sfdupes/milestone/17 — the milestone is the source
|
||||
of truth for what is left before 1.0.0. Individual issues are deliberately not
|
||||
restated here; a copy in this file drifts out of date the moment the tracker
|
||||
moves
|
||||
https://git.eeqj.de/sneak/sfdupes/milestone/17 — the milestone is
|
||||
the source of truth for what is left before 1.0.0. Individual
|
||||
issues are deliberately not restated here; a copy in this file
|
||||
drifts out of date the moment the tracker moves
|
||||
|
||||
# Completed Steps
|
||||
|
||||
- restore Markdown formatting in `script/fmt`/`fmt-check` and reformat all
|
||||
Markdown to the house prettier settings (2026-09-21, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/19)
|
||||
- stamp the git tag or short commit in a plain `docker build .`
|
||||
instead of `dev` (2026-10-02, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now
|
||||
sends `.git`, without `.git/config`, and the `Dockerfile` build
|
||||
stage takes the `VERSION` build argument when one is given,
|
||||
otherwise `git describe --tags --always` of that `.git`. The build
|
||||
fails if the context carries `.git` and the version still comes out
|
||||
empty, `dev` or `unknown`. The CI checkout step fetches the full
|
||||
history (`fetch-depth: 0`) so CI sees the tag and stamps the same
|
||||
value as `make build`.
|
||||
|
||||
- replace the 1 KiB end-window sampling with the head/tail plus
|
||||
content-hash ladder (2026-09-22, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is
|
||||
hashed in full and compared directly, with no end-window step — its
|
||||
`head`, `tail`, and `content` all hold the whole-file hash. A file at
|
||||
10 MiB or above gets only the 64 KiB `head` and `tail` in the hash
|
||||
phase; a new content phase, after the update phase, reads it for its
|
||||
`content` hash — the whole file below 50 MiB, gigabyte-spaced 1 MiB
|
||||
samples at or above — only when its size, `head`, and `tail` match
|
||||
another record's, from the same scan or stored by an earlier one, so
|
||||
a stored file gains its content hash when it gains a match. A file
|
||||
that is gone or has changed since its record was written is not
|
||||
read. The `content` column is part of the version 1 schema. `report`
|
||||
and `trees` group by the extended signature and leave out any record
|
||||
without a `content` hash, so the ladder is applied across the whole
|
||||
database. README "Duplicate detection" documents every rung including
|
||||
the probabilistic large-file path.
|
||||
|
||||
- remove the dead `files.dat` references from `Makefile`, `.gitignore`
|
||||
and `.dockerignore` (2026-09-21, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/22)
|
||||
|
||||
- fix the lint-image pin comments and `FROM` form in `Dockerfile` and
|
||||
`Dockerfile.lint` (2026-08-10, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/25): dropped the false
|
||||
`(Debian-based)` parenthetical (v2.12.1 was Debian too) and the redundant tag,
|
||||
so both pins are the policy `# image:vX.Y.Z, YYYY-MM-DD` comment over a bare
|
||||
`FROM image@sha256:...`. Digest unchanged. `script/verify-lint-image-pin`
|
||||
parses those `FROM` lines and still matches the tagless form; its advice line
|
||||
lost the now meaningless "tag and digest". With no tag in either reference, a
|
||||
tag-only disagreement no longer exists — a one-sided tag is caught as a plain
|
||||
mismatch.
|
||||
`(Debian-based)` parenthetical (v2.12.1 was Debian too) and the
|
||||
redundant tag, so both pins are the policy `# image:vX.Y.Z,
|
||||
YYYY-MM-DD` comment over a bare `FROM image@sha256:...`. Digest
|
||||
unchanged. `script/verify-lint-image-pin` parses those `FROM` lines
|
||||
and still matches the tagless form; its advice line lost the now
|
||||
meaningless "tag and digest". With no tag in either reference, a
|
||||
tag-only disagreement no longer exists — a one-sided tag is caught as
|
||||
a plain mismatch.
|
||||
|
||||
- run all linting in Docker via `Dockerfile.lint` and `script/lint` (2026-08-10,
|
||||
branch `next`, closes https://git.eeqj.de/sneak/sfdupes/issues/46): per the
|
||||
owner ruling, the linter runs inside a container invoked through the `script/`
|
||||
entrypoint and is never installed on a host. New root `Dockerfile.lint` COPYs
|
||||
the repo into the digest-pinned `golangci/golangci-lint:v2.12.2` image and
|
||||
runs `golangci-lint config verify` and `golangci-lint run` as build steps, so
|
||||
a successful build IS a clean lint; `script/lint` is reduced to building it.
|
||||
`script/bootstrap` loses the `go install`, the pin constants, the version
|
||||
parser and `verify_golangci_lint` outright rather than hardening them — with
|
||||
nothing linting on the host, the `$GOPATH/bin` versus `PATH` problem that
|
||||
motivated them has no subject — and now warns rather than fails when `docker`
|
||||
is absent. Two traps handled. A lint build on an unchanged tree returns
|
||||
success in well under a second having run no linter, which is
|
||||
- run all linting in Docker via `Dockerfile.lint` and `script/lint`
|
||||
(2026-08-10, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/46): per the owner ruling, the
|
||||
linter runs inside a container invoked through the `script/`
|
||||
entrypoint and is never installed on a host. New root
|
||||
`Dockerfile.lint` COPYs the repo into the digest-pinned
|
||||
`golangci/golangci-lint:v2.12.2` image and runs
|
||||
`golangci-lint config verify` and `golangci-lint run` as build
|
||||
steps, so a successful build IS a clean lint; `script/lint` is
|
||||
reduced to building it. `script/bootstrap` loses the `go install`,
|
||||
the pin constants, the version parser and `verify_golangci_lint`
|
||||
outright rather than hardening them — with nothing linting on the
|
||||
host, the `$GOPATH/bin` versus `PATH` problem that motivated them has
|
||||
no subject — and now warns rather than fails when `docker` is absent.
|
||||
Two traps handled. A lint build on an unchanged tree returns success
|
||||
in well under a second having run no linter, which is
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/32 and
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/39 again, so `Dockerfile.lint`
|
||||
carries `ARG CHECK_EPOCH` referenced inside every gate `RUN` (BuildKit hashes
|
||||
the expanded command, not the declaration) and `script/lint` passes
|
||||
`"$(date +%s)-$$"` — the PID matters because two lint runs land inside the
|
||||
same second easily. And nothing inside an image build may shell out to docker,
|
||||
so the main `Dockerfile`'s lint stage now invokes `golangci-lint` directly
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/39 again, so
|
||||
`Dockerfile.lint` carries `ARG CHECK_EPOCH` referenced
|
||||
inside every gate `RUN` (BuildKit hashes the expanded command, not
|
||||
the declaration) and `script/lint` passes `"$(date +%s)-$$"` — the
|
||||
PID matters because two lint runs land inside the same second easily.
|
||||
And nothing inside an image build may shell out to docker, so the
|
||||
main `Dockerfile`'s lint stage now invokes `golangci-lint` directly
|
||||
instead of `make lint`, and its build stage runs `make test` and
|
||||
`make fmt-check` instead of the `make check` aggregate (`make`, not the
|
||||
scripts bare, because the Makefile's `export CGO_ENABLED = 0` only reaches
|
||||
what it invokes). `COPY --from=lint` `/usr/bin/golangci-lint` is replaced by
|
||||
`COPY --from=lint /src/go.sum /dev/null`: the copied binary was the only edge
|
||||
forcing BuildKit to finish linting before the build stage starts, and dropping
|
||||
it without replacing the edge would have ended fail-fast linting silently
|
||||
under a still-green build. That is canonical `REPO_POLICIES.md:107`'s ordering
|
||||
edge, restored. `ENV PATH=/home/builder/go/bin:$PATH` is gone with the
|
||||
`go install` that justified it. `script/verify-linter-pin` is retired, deleted
|
||||
along with its README entry, because both of its subjects ceased to exist in
|
||||
the same change: it compared a linter binary against `GOLANGCI_LINT_VERSION`
|
||||
in `script/bootstrap`, and there is now neither a binary crossing between
|
||||
stages nor a version pin in bootstrap. The drift it guarded has not gone away,
|
||||
it has moved — the linter is still pinned twice, now as the `FROM` line of
|
||||
`Dockerfile.lint` and the `FROM` line of the `Dockerfile` lint stage, with
|
||||
nothing syncing them, which is exactly what
|
||||
`make fmt-check` instead of the `make check` aggregate (`make`, not
|
||||
the scripts bare, because the Makefile's `export CGO_ENABLED = 0`
|
||||
only reaches what it invokes). `COPY --from=lint`
|
||||
`/usr/bin/golangci-lint` is replaced by
|
||||
`COPY --from=lint /src/go.sum /dev/null`: the copied binary was the
|
||||
only edge forcing BuildKit to finish linting before the build stage
|
||||
starts, and dropping it without replacing the edge would have ended
|
||||
fail-fast linting silently under a still-green build. That is
|
||||
canonical `REPO_POLICIES.md:107`'s ordering edge, restored.
|
||||
`ENV PATH=/home/builder/go/bin:$PATH` is gone with the `go install`
|
||||
that justified it. `script/verify-linter-pin` is retired, deleted
|
||||
along with its README entry, because both of its subjects ceased to
|
||||
exist in the same change: it compared a linter binary against
|
||||
`GOLANGCI_LINT_VERSION` in `script/bootstrap`, and there is now
|
||||
neither a binary crossing between stages nor a version pin in
|
||||
bootstrap. The drift it guarded has not gone away, it has moved — the
|
||||
linter is still pinned twice, now as the `FROM` line of
|
||||
`Dockerfile.lint` and the `FROM` line of the `Dockerfile` lint stage,
|
||||
with nothing syncing them, which is exactly what
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/42 made a build failure. Its
|
||||
replacement is one new `script/verify-lint-image-pin`, run as a gate in both
|
||||
files, which compares the two references to each other and deliberately
|
||||
restates neither: a hardcoded expected digest would be a third copy and the
|
||||
same drift one file further out. `golangci-lint config verify` is included per
|
||||
the ruling, and the concern about its unpinned live HTTPS schema fetch was
|
||||
measured rather than assumed — under `--network none` the pinned binary both
|
||||
passes a valid config and rejects an invalid one with the jsonschema error, so
|
||||
it validates from an embedded schema and makes no network call of its own. The
|
||||
README scopes that to the gate steps rather than to linting as a whole:
|
||||
`Dockerfile.lint` runs `go mod download` above them, so a cold cache still
|
||||
needs the network and only a warm one lints offline. Verified: `make lint`
|
||||
green with every `PATH` directory containing a `golangci-lint` removed
|
||||
replacement is one new `script/verify-lint-image-pin`,
|
||||
run as a gate in both files, which compares the two references to
|
||||
each other and deliberately restates neither: a hardcoded expected
|
||||
digest would be a third copy and the same drift one file further out.
|
||||
`golangci-lint config verify` is included per the ruling, and the
|
||||
concern about its unpinned live HTTPS schema fetch was measured
|
||||
rather than assumed — under `--network none` the pinned binary both
|
||||
passes a valid config and rejects an invalid one with the jsonschema
|
||||
error, so it validates from an embedded schema and makes no network
|
||||
call of its own. The README scopes that to the gate steps rather
|
||||
than to linting as a whole: `Dockerfile.lint` runs `go mod download`
|
||||
above them, so a cold cache still needs the network and only a warm
|
||||
one lints offline. Verified: `make lint` green with every `PATH`
|
||||
directory containing a `golangci-lint` removed
|
||||
(`/home/user/go/bin`, `/home/user/.local/bin`, `/usr/local/bin`;
|
||||
`command -v golangci-lint` empty); two consecutive `script/lint` runs on an
|
||||
untouched tree both executed the linter, 27.7s and 28.7s in the lint step
|
||||
under distinct epochs with the `COPY . .` layer `CACHED` above them, at 42.2s
|
||||
and 41.8s wall clock — the no-cache rule was not weakened to shorten that.
|
||||
Negative control: a planted `var unusedIssue46Sentinel = 1` failed
|
||||
`script/lint` with
|
||||
`report.go:173:5: var unusedIssue46Sentinel is unused (unused)`, and failed
|
||||
`make docker` at `[lint 9/9]` with the build stage stopped at `[builder 3/12]`
|
||||
— `COPY --from=lint`, `script/bootstrap`, the test gate and `make build` all
|
||||
zero occurrences — then reverted clean. The drift guard fails on a tag-only
|
||||
disagreement, on a digest-only disagreement, and on an unreadable reference,
|
||||
naming both sides. `make docker` green in 5m35s with all six gates executing
|
||||
under one epoch (lint 37.6s, test 25.2s reporting
|
||||
`ok sneak.berlin/go/sfdupes 1.938s coverage: 88.5%`, not `(cached)`). The
|
||||
non-root quirk still holds: in the builder image with the Go test cache off,
|
||||
`--user 0:0` fails `TestScanHardlinkRunFailsTogether` (exit 1) where the
|
||||
unprivileged user passes (exit 0). Noted for follow-up, not fixed here:
|
||||
`golangci-lint` warns that the `gomodguard` linter is deprecated since v2.12.0
|
||||
in favour of `gomodguard_v2`.
|
||||
`command -v golangci-lint` empty); two consecutive `script/lint` runs
|
||||
on an untouched tree both executed the linter, 27.7s and 28.7s in the
|
||||
lint step under distinct epochs with the `COPY . .` layer `CACHED`
|
||||
above them, at 42.2s and 41.8s wall clock — the no-cache rule was not
|
||||
weakened to shorten that. Negative control: a planted
|
||||
`var unusedIssue46Sentinel = 1` failed `script/lint` with
|
||||
`report.go:173:5: var unusedIssue46Sentinel is unused (unused)`, and
|
||||
failed `make docker` at `[lint 9/9]` with the build stage stopped at
|
||||
`[builder 3/12]` — `COPY --from=lint`, `script/bootstrap`, the test
|
||||
gate and `make build` all zero occurrences — then reverted clean. The
|
||||
drift guard fails on a tag-only disagreement, on a digest-only
|
||||
disagreement, and on an unreadable reference, naming both sides.
|
||||
`make docker` green in 5m35s with all six gates executing under one
|
||||
epoch (lint 37.6s, test 25.2s reporting
|
||||
`ok sneak.berlin/go/sfdupes 1.938s coverage: 88.5%`, not `(cached)`).
|
||||
The non-root quirk still holds: in the builder image with the Go test
|
||||
cache off, `--user 0:0` fails `TestScanHardlinkRunFailsTogether`
|
||||
(exit 1) where the unprivileged user passes (exit 0). Noted for
|
||||
follow-up, not fixed here: `golangci-lint` warns that the
|
||||
`gomodguard` linter is deprecated since v2.12.0 in favour of
|
||||
`gomodguard_v2`.
|
||||
|
||||
- install the Docker build stage's prerequisites by running `script/bootstrap`
|
||||
instead of `apk add --no-cache make` inline (2026-08-09, branch
|
||||
`dockerfile-bootstrap`, closes #42): canonical `REPO_POLICIES.md:97` requires
|
||||
it, and the inline install left the build stage maintaining its own notion of
|
||||
the toolchain — exactly the divergence #24 exists to close, one layer down.
|
||||
The stage now copies `script/` plus `go.mod`/`go.sum` and runs
|
||||
`script/bootstrap`, which ends in `go mod download`, so the separate
|
||||
invocation of that is gone. `COPY --from=lint /usr/bin/golangci-lint` stays,
|
||||
and moves above the bootstrap layer. It is the only edge making this stage
|
||||
depend on the lint stage, so deleting it as redundant would end fail-fast
|
||||
linting silently. Letting bootstrap install its own linter here would have
|
||||
reintroduced the second toolchain and paid for a from-source build of it. What
|
||||
makes the two stages provably one toolchain rather than two that happen to
|
||||
agree is a new `script/verify-linter-pin`, run in the build stage on the
|
||||
binary that arrives from the lint stage, before bootstrap: it fails the build
|
||||
naming both versions unless that binary is the version `script/bootstrap`
|
||||
pins. Bootstrap's own check could not serve that purpose — it reinstalls its
|
||||
pin from source and then verifies whatever `PATH` resolves, so drift
|
||||
self-heals silently and a lint stage image bumped on its own would lint at the
|
||||
new version while `make check` ran at the old one, green. The linter version
|
||||
is pinned in two independent places (the lint stage image digest and
|
||||
- install the Docker build stage's prerequisites by running
|
||||
`script/bootstrap` instead of `apk add --no-cache make` inline
|
||||
(2026-08-09, branch `dockerfile-bootstrap`, closes #42): canonical
|
||||
`REPO_POLICIES.md:97` requires it, and the inline install left the
|
||||
build stage maintaining its own notion of the toolchain — exactly
|
||||
the divergence #24 exists to close, one layer down. The stage now
|
||||
copies `script/` plus `go.mod`/`go.sum` and runs `script/bootstrap`,
|
||||
which ends in `go mod download`, so the separate invocation of that
|
||||
is gone. `COPY --from=lint /usr/bin/golangci-lint` stays, and moves
|
||||
above the bootstrap layer. It is the only edge making this stage
|
||||
depend on the lint stage, so deleting it as redundant would end
|
||||
fail-fast linting silently. Letting bootstrap install its own linter
|
||||
here would have reintroduced the second toolchain and paid for a
|
||||
from-source build of it. What makes the two stages provably one
|
||||
toolchain rather than two that happen to agree is a new
|
||||
`script/verify-linter-pin`, run in the build stage on the binary
|
||||
that arrives from the lint stage, before bootstrap: it fails the
|
||||
build naming both versions unless that binary is the version
|
||||
`script/bootstrap` pins. Bootstrap's own check could not serve that
|
||||
purpose — it reinstalls its pin from source and then verifies
|
||||
whatever `PATH` resolves, so drift self-heals silently and a lint
|
||||
stage image bumped on its own would lint at the new version while
|
||||
`make check` ran at the old one, green. The linter version is pinned
|
||||
in two independent places (the lint stage image digest and
|
||||
`GOLANGCI_LINT_VERSION`) and nothing else keeps them in sync, so a
|
||||
half-applied bump is now a build failure. The pin is read out of
|
||||
`script/bootstrap`, which stays the single source of truth; a pin that cannot
|
||||
be read is a hard failure, not a skip. The check needs no `CHECK_EPOCH`: its
|
||||
only inputs are the copied binary and `script/`, so Docker invalidates the
|
||||
layer exactly when a cached result would stop being true, and it is documented
|
||||
with the other entrypoints in the README. `$GOPATH/bin` joins `PATH` because
|
||||
that is where bootstrap's `go install` lands and bootstrap verifies its
|
||||
installs against what `PATH` resolves — nothing in the image is shadowed by
|
||||
it, the directory does not exist until bootstrap runs. Everything added sits
|
||||
above `ARG CHECK_EPOCH`, and the `chown` and `USER builder` still precede
|
||||
`make check`. Verified: the guard fails the build with both versions named
|
||||
when the lint stage's linter is faked to a different version, and an
|
||||
unmodified build still passes it; bootstrap runs clean under Alpine's `sh` and
|
||||
its `apk` branch, installing `git` and `make` and finding the copied linter
|
||||
already at the pin; a second build served the bootstrap and dependency layers
|
||||
`CACHED` while both gates ran with a fresh epoch; a planted `unused` finding
|
||||
failed the build at the lint gate in 48.9s with the build stage's `make check`
|
||||
never starting; and the suite run in the image as `--user 0:0` fails
|
||||
`TestScanHardlinkRunFailsTogether`, so the drop to the unprivileged user is
|
||||
still load-bearing. That last check needs the Go test cache disabled — the
|
||||
first attempt reported `ok ... (cached)` as root, reusing the result the
|
||||
build-time run had left in the shared cache, which would have read as a pass.
|
||||
Build wall time, on a shared host running many concurrent builds and so noisy:
|
||||
2m13s on an unchanged tree, 2m17s and 4m29s for two builds after a source
|
||||
change, 5m14s cold. Only the cold one breaches the policy ceiling, and not
|
||||
because of this change — `chown -R builder:builder /src /home/builder` walks
|
||||
the module cache and re-runs on every source change, and it alone varied
|
||||
between 77s and 210s across those four builds, which is also the whole spread
|
||||
in the totals. The same cold measurement against `main` is 5m03s with a 209s
|
||||
`chown`. Filed as #43
|
||||
- bust the Docker layer cache for the gate steps, so `script/cibuild` and
|
||||
`script/docker` cannot report a green they did not earn (2026-08-09, branch
|
||||
`cibuild-cache-bust`, closes #32): both scripts were bare `docker build`
|
||||
invocations with no cache control, and the `Dockerfile` copies the tree before
|
||||
running its gates, so on an unchanged tree Docker served those layers from
|
||||
cache and the build exited 0 having executed nothing. That is not hypothetical
|
||||
here — every merge this repo has done is a non-fast-forward merge of an
|
||||
undiverged branch, so each merge commit's tree is byte-identical to the branch
|
||||
head's and each merge CI run was almost certainly a full cache hit; and PR
|
||||
#31's reviewer found `make docker` returning success as a 17-layer cache hit,
|
||||
catching it only by being suspicious. The fix is `ARG CHECK_EPOCH` with the
|
||||
scripts passing `--build-arg CHECK_EPOCH="$(date +%s)"`. Two details make or
|
||||
break it. `ARG` is scoped per stage and this `Dockerfile` has three gates
|
||||
across two — `make fmt-check` and `make lint` in the lint stage, `make check`
|
||||
in the build stage — so a single declaration would have left one stage
|
||||
silently cacheable; it is declared in both. And BuildKit hashes the expanded
|
||||
command, not the declaration, so a declared-but-unreferenced `ARG` invalidates
|
||||
nothing: each gate `RUN` echoes the epoch, which also puts the value in the
|
||||
build log as evidence the layer really ran. Placement is below the dependency
|
||||
layers on purpose — a build that goes cold every time would be a different
|
||||
bug, not a fix. Verified by running each script twice back to back on an
|
||||
unchanged tree under `BUILDKIT_PROGRESS=plain`: all three gates executed on
|
||||
all four runs, each with a fresh epoch in the log (`script/cibuild` 78.8s then
|
||||
61.1s; `script/docker` 61.1s then 53.4s), and twelve steps were still served
|
||||
`CACHED` in the steady state — both `go mod download`s, `apk add`, `adduser`,
|
||||
the `chown`, every `go.mod`/`go.sum` and source copy, the linter copy out of
|
||||
the lint stage, and the binary copy into the runtime stage. The lint stage
|
||||
still gates the build stage: with a deliberate `unused` finding planted in the
|
||||
tree, the build failed at `make lint` in 36.1s and the build-stage
|
||||
`make check` never started. The build stage also still drops to the
|
||||
unprivileged `builder` user before `make check`, which the suite depends on
|
||||
rather than merely prefers: forcing the same image to run the tests as root
|
||||
fails `TestScanHardlinkRunFailsTogether`, because root reads straight through
|
||||
the `chmod(0)` the test uses to prove hard links are read once. This is the
|
||||
local fix only; propagating it to the canonical templates is `prompts` #26
|
||||
- check the installed golangci-lint version in `script/bootstrap` instead of
|
||||
only its presence (2026-08-09, branch `bootstrap-version-check`, closes #24):
|
||||
`missing golangci-lint` meant any linter already on `PATH` satisfied the
|
||||
check, so the pin was never consulted and the v2.12.2 bump from #3 was inert
|
||||
on every host that already had one — this host ran v2.10.1 against a v2.12.2
|
||||
pin, `make check` went green, and `make docker` then rejected the same commit
|
||||
with findings the local gate never saw. The version now lives in one place,
|
||||
`GOLANGCI_LINT_VERSION`, with the `go install` module ref derived from it so a
|
||||
bump cannot half-apply; a `golangci_lint_version` helper parses
|
||||
`golangci-lint --version` (taking the field after the word `version` and
|
||||
tolerating an optional leading `v`, which the module ref carries and the
|
||||
binary's output does not), and any version that is not the pin — older, newer,
|
||||
absent or unparseable — is reinstalled. The install is then verified against
|
||||
the binary `PATH` actually resolves: `go install` writes into `GOBIN` (or
|
||||
`GOPATH/bin`) while `make lint` runs whichever `golangci-lint` comes first on
|
||||
`PATH`, so a wrong-version one sitting ahead of it — nix, apt, brew, apk, or
|
||||
the `/usr/local/bin` copy the `Dockerfile` builder stage makes — would swallow
|
||||
the install and leave the local gate disagreeing with CI under an affirmative
|
||||
`bootstrap complete`. Bootstrap now re-reads the effective version after
|
||||
installing and, on a mismatch, prints both paths and both versions to stderr
|
||||
and exits non-zero instead of claiming success; it does not reorder anyone's
|
||||
`script/bootstrap`, which stays the single source of truth; a pin
|
||||
that cannot be read is a hard failure, not a skip. The check needs
|
||||
no `CHECK_EPOCH`: its only inputs are the copied binary and
|
||||
`script/`, so Docker invalidates the layer exactly when a cached
|
||||
result would stop being true, and it is documented with the other
|
||||
entrypoints in the README. `$GOPATH/bin` joins `PATH` because
|
||||
that is where bootstrap's `go install` lands and bootstrap verifies
|
||||
its installs against what `PATH` resolves — nothing in the image is
|
||||
shadowed by it, the directory does not exist until bootstrap runs.
|
||||
Everything added sits above `ARG CHECK_EPOCH`, and the `chown` and
|
||||
`USER builder` still precede `make check`. Verified: the guard fails
|
||||
the build with both versions named when the lint stage's linter is
|
||||
faked to a different version, and an unmodified build still passes
|
||||
it; bootstrap runs clean under Alpine's `sh` and its `apk` branch,
|
||||
installing `git` and `make` and finding the copied
|
||||
linter already at the pin; a second build served the bootstrap and
|
||||
dependency layers `CACHED` while both gates ran with a fresh epoch;
|
||||
a planted `unused` finding failed the build at the lint gate in
|
||||
48.9s with the build stage's `make check` never starting; and the
|
||||
suite run in the image as `--user 0:0` fails
|
||||
`TestScanHardlinkRunFailsTogether`, so the drop to the unprivileged
|
||||
user is still load-bearing. That last check needs the Go test cache
|
||||
disabled — the first attempt reported `ok ... (cached)` as root,
|
||||
reusing the result the build-time run had left in the shared cache,
|
||||
which would have read as a pass. Build wall time, on a shared host
|
||||
running many concurrent builds and so noisy: 2m13s on an unchanged
|
||||
tree, 2m17s and 4m29s for two builds after a source change, 5m14s
|
||||
cold. Only the cold one breaches the policy ceiling, and not because
|
||||
of this change — `chown -R builder:builder /src /home/builder` walks
|
||||
the module cache and re-runs on every source change, and it alone
|
||||
varied between 77s and 210s across those four builds, which is also
|
||||
the whole spread in the totals. The same cold measurement against
|
||||
`main` is 5m03s with a 209s `chown`. Filed as #43
|
||||
- bust the Docker layer cache for the gate steps, so `script/cibuild`
|
||||
and `script/docker` cannot report a green they did not earn
|
||||
(2026-08-09, branch `cibuild-cache-bust`, closes #32): both scripts
|
||||
were bare `docker build` invocations with no cache control, and the
|
||||
`Dockerfile` copies the tree before running its gates, so on an
|
||||
unchanged tree Docker served those layers from cache and the build
|
||||
exited 0 having executed nothing. That is not hypothetical here —
|
||||
every merge this repo has done is a non-fast-forward merge of an
|
||||
undiverged branch, so each merge commit's tree is byte-identical to
|
||||
the branch head's and each merge CI run was almost certainly a full
|
||||
cache hit; and PR #31's reviewer found `make docker` returning
|
||||
success as a 17-layer cache hit, catching it only by being
|
||||
suspicious. The fix is `ARG CHECK_EPOCH` with the scripts passing
|
||||
`--build-arg CHECK_EPOCH="$(date +%s)"`. Two details make or break
|
||||
it. `ARG` is scoped per stage and this `Dockerfile` has three gates
|
||||
across two — `make fmt-check` and `make lint` in the lint stage,
|
||||
`make check` in the build stage — so a single declaration would have
|
||||
left one stage silently cacheable; it is declared in both. And
|
||||
BuildKit hashes the expanded command, not the declaration, so a
|
||||
declared-but-unreferenced `ARG` invalidates nothing: each gate `RUN`
|
||||
echoes the epoch, which also puts the value in the build log as
|
||||
evidence the layer really ran. Placement is below the dependency
|
||||
layers on purpose — a build that goes cold every time would be a
|
||||
different bug, not a fix. Verified by running each script twice back
|
||||
to back on an unchanged tree under `BUILDKIT_PROGRESS=plain`: all
|
||||
three gates executed on all four runs, each with a fresh epoch in
|
||||
the log (`script/cibuild` 78.8s then 61.1s; `script/docker` 61.1s
|
||||
then 53.4s), and twelve steps were still served `CACHED` in the
|
||||
steady state — both `go mod download`s, `apk add`, `adduser`, the
|
||||
`chown`, every `go.mod`/`go.sum` and source copy, the linter copy
|
||||
out of the lint stage, and the binary copy into the runtime stage.
|
||||
The lint stage still gates the build stage: with a deliberate
|
||||
`unused` finding planted in the tree, the build failed at
|
||||
`make lint` in 36.1s and the build-stage `make check` never started.
|
||||
The build stage also still drops to the unprivileged `builder` user
|
||||
before `make check`, which the suite depends on rather than merely
|
||||
prefers: forcing the same image to run the tests as root fails
|
||||
`TestScanHardlinkRunFailsTogether`, because root reads straight
|
||||
through the `chmod(0)` the test uses to prove hard links are read
|
||||
once. This is the local fix only; propagating it to the canonical
|
||||
templates is `prompts` #26
|
||||
- check the installed golangci-lint version in `script/bootstrap`
|
||||
instead of only its presence (2026-08-09, branch
|
||||
`bootstrap-version-check`, closes #24): `missing golangci-lint` meant
|
||||
any linter already on `PATH` satisfied the check, so the pin was never
|
||||
consulted and the v2.12.2 bump from #3 was inert on every host that
|
||||
already had one — this host ran v2.10.1 against a v2.12.2 pin,
|
||||
`make check` went green, and `make docker` then rejected the same
|
||||
commit with findings the local gate never saw. The version now lives
|
||||
in one place, `GOLANGCI_LINT_VERSION`, with the `go install` module
|
||||
ref derived from it so a bump cannot half-apply; a
|
||||
`golangci_lint_version` helper parses `golangci-lint --version`
|
||||
(taking the field after the word `version` and tolerating an optional
|
||||
leading `v`, which the module ref carries and the binary's output does
|
||||
not), and any version that is not the pin — older, newer, absent or
|
||||
unparseable — is reinstalled. The install is then verified against the
|
||||
binary `PATH` actually resolves: `go install` writes into `GOBIN` (or
|
||||
`GOPATH/bin`) while `make lint` runs whichever `golangci-lint` comes
|
||||
first on `PATH`, so a wrong-version one sitting ahead of it — nix,
|
||||
apt, brew, apk, or the `/usr/local/bin` copy the `Dockerfile` builder
|
||||
stage makes — would swallow the install and leave the local gate
|
||||
disagreeing with CI under an affirmative `bootstrap complete`.
|
||||
Bootstrap now re-reads the effective version after installing and, on
|
||||
a mismatch, prints both paths and both versions to stderr and exits
|
||||
non-zero instead of claiming success; it does not reorder anyone's
|
||||
`PATH` or delete their binary. The `--version` call keeps its stderr
|
||||
connected, so a present-but-broken binary says why rather than reinstalling
|
||||
forever in silence, and is bounded by `timeout(1)` where that exists, so a
|
||||
wedged binary cannot hang bootstrap. `git`, `make` and `go` keep their
|
||||
presence-only checks and now say why in a comment: they are host
|
||||
package-manager tools the repo deliberately does not pin, with `go.mod`
|
||||
governing the language version and the digest-pinned images covering
|
||||
reproducible builds. Verified on this host by bootstrapping from v2.10.1 to
|
||||
v2.12.2 and running it again to a no-op, plus stub runs of the real script
|
||||
under `dash` covering a thirteen-input parse matrix (absent, older, newer,
|
||||
host-style, image-style, leading-`v`, stderr-only, empty, non-zero exit,
|
||||
impostor binary, `(devel)`, trailing `version`), a shadowed install that must
|
||||
exit non-zero, an install destination not on `PATH` at all, `GOBIN` set, and a
|
||||
wedged binary that must hit the timeout; `make check` and `make lint` are
|
||||
clean at v2.12.2, so v2.10.1 was not hiding any findings on `main`
|
||||
connected, so a present-but-broken binary says why rather than
|
||||
reinstalling forever in silence, and is bounded by `timeout(1)` where
|
||||
that exists, so a wedged binary cannot hang bootstrap. `git`, `make`
|
||||
and `go` keep their presence-only checks and now say why in a
|
||||
comment: they are host package-manager tools the repo deliberately
|
||||
does not pin, with `go.mod` governing the language version and the
|
||||
digest-pinned images covering reproducible builds. Verified on this
|
||||
host by bootstrapping from v2.10.1 to v2.12.2 and running it again to
|
||||
a no-op, plus stub runs of the real script under `dash` covering a
|
||||
thirteen-input parse matrix (absent, older, newer, host-style,
|
||||
image-style, leading-`v`, stderr-only, empty, non-zero exit, impostor
|
||||
binary, `(devel)`, trailing `version`), a shadowed install that must
|
||||
exit non-zero, an install destination not on `PATH` at all, `GOBIN`
|
||||
set, and a wedged binary that must hit the timeout; `make check` and
|
||||
`make lint` are clean at v2.12.2, so v2.10.1 was not hiding any
|
||||
findings on `main`
|
||||
- unwind the hash worker pool on the error path (2026-08-09, branch
|
||||
`hash-pool-cleanup`, closes #6): `hashPhase` used to return the moment
|
||||
`recordRun` failed and abandon the pool — the feeder parked forever on a full
|
||||
`jobs` channel and every worker on a full `results` channel. That only stopped
|
||||
being invisible when #4 landed and `runScan` began unwinding instead of
|
||||
calling `os.Exit`. The pool is now an owned, context-aware `hashPool`: every
|
||||
blocking send in the feeder and the workers selects on `ctx.Done()`, `jobs` is
|
||||
closed on every path out, and `hashPhase` defers `pool.stop()`, which cancels
|
||||
and then drains `results` until the last goroutine has exited — draining is
|
||||
what frees a worker already parked on a send. `ctx` is threaded from
|
||||
`cmd.Context()` through `runScan`, `syncScan`, both worker pools and the whole
|
||||
database layer (it is the first parameter everywhere), so #5 can hand this
|
||||
path a signal and needs to add nothing else. The walk pool never leaked,
|
||||
because `walkPhase` always drains its events to close, but it has the same
|
||||
unbounded-send shape and #5 will give it an early return, so it gets the same
|
||||
treatment plus a `ctx.Err()` guard after the walk: a cancelled walk yields a
|
||||
partial size census, and every file it never reached looks vanished to the
|
||||
update phase. That phase's own `BeginTx` fails on the same cancelled context
|
||||
before deleting anything, so the guard is defence in depth rather than the
|
||||
only barrier — but it is the one that survives #5 deciding an interrupted scan
|
||||
may commit what it has. Tests drive `run(scan)` against a database whose
|
||||
insert trigger aborts, and assert both that the scan fails instead of hanging
|
||||
and that `runtime.NumGoroutine()` polls back to its pre-scan baseline; a
|
||||
second set cancels a scan part-way through the walk — deterministically, by
|
||||
counting the scan's own consultations of `ctx.Done()` rather than racing a
|
||||
timer — and asserts that it stops at the guard holding a partial census and a
|
||||
still-populated record index, with every record intact. The remaining
|
||||
cancellation branches of both pools are covered by direct tests of
|
||||
`sendEvent`, the walk workers, `dispatchDirs`, `feedHashJobs`, `hashWorker`
|
||||
and `hashPhase`
|
||||
- guarantee the database is closed on every fatal exit path (2026-08-09, branch
|
||||
`db-close-on-fatal`, closes #4): `fatalf` and its `os.Exit(1)` are gone, so
|
||||
the deferred `db.Close()` — and with it the SQLite WAL checkpoint — now
|
||||
actually runs when a subcommand fails; `runScan`, `runReport`, `runTrees`,
|
||||
`loadRecords` and `resolveRoots` return errors instead. The single exit point
|
||||
is `run` in `main.go`: it maps a `fatalError` (anything a subcommand returned)
|
||||
to exit 1 and cobra's own argument and flag errors to exit 2, which keeps a
|
||||
runtime failure from being reported as a usage error or printing the usage
|
||||
text. New `main_test.go` drives the CLI in-process and asserts the exit codes
|
||||
from README §Error handling plus the stdout/stderr split, including that a
|
||||
fatal error raised after the database is open leaves no `-wal`/`-shm` sidecar
|
||||
behind for `scan`, `report` or `trees`
|
||||
- update golangci-lint to v2.12.2 with the canonical config (2026-08-09, branch
|
||||
`golangci-v2.12.2`, merged as `38a01bd`, closes #3): bumped the pinned linter
|
||||
in the `Dockerfile` lint stage and `script/bootstrap` from v2.12.1 to v2.12.2,
|
||||
and replaced `.golangci.yml` with the canonical file — the linter settings
|
||||
`hash-pool-cleanup`, closes #6): `hashPhase` used to return the
|
||||
moment `recordRun` failed and abandon the pool — the feeder parked
|
||||
forever on a full `jobs` channel and every worker on a full
|
||||
`results` channel. That only stopped being invisible when #4 landed
|
||||
and `runScan` began unwinding instead of calling `os.Exit`. The
|
||||
pool is now an owned, context-aware `hashPool`: every blocking send
|
||||
in the feeder and the workers selects on `ctx.Done()`, `jobs` is
|
||||
closed on every path out, and `hashPhase` defers `pool.stop()`,
|
||||
which cancels and then drains `results` until the last goroutine
|
||||
has exited — draining is what frees a worker already parked on a
|
||||
send. `ctx` is threaded from `cmd.Context()` through `runScan`,
|
||||
`syncScan`, both worker pools and the whole database layer (it is
|
||||
the first parameter everywhere), so #5 can hand this path a signal
|
||||
and needs to add nothing else. The walk pool never leaked, because
|
||||
`walkPhase` always drains its events to close, but it has the same
|
||||
unbounded-send shape and #5 will give it an early return, so it
|
||||
gets the same treatment plus a `ctx.Err()` guard after the walk: a
|
||||
cancelled walk yields a partial size census, and every file it never
|
||||
reached looks vanished to the update phase. That phase's own
|
||||
`BeginTx` fails on the same cancelled context before deleting
|
||||
anything, so the guard is defence in depth rather than the only
|
||||
barrier — but it is the one that survives #5 deciding an interrupted
|
||||
scan may commit what it has. Tests drive `run(scan)` against a
|
||||
database whose insert trigger aborts, and assert both that the scan
|
||||
fails instead of hanging and that `runtime.NumGoroutine()` polls
|
||||
back to its pre-scan baseline; a second set cancels a scan part-way
|
||||
through the walk — deterministically, by counting the scan's own
|
||||
consultations of `ctx.Done()` rather than racing a timer — and
|
||||
asserts that it stops at the guard holding a partial census and a
|
||||
still-populated record index, with every record intact. The
|
||||
remaining cancellation branches of both pools are covered by direct
|
||||
tests of `sendEvent`, the walk workers, `dispatchDirs`,
|
||||
`feedHashJobs`, `hashWorker` and `hashPhase`
|
||||
- guarantee the database is closed on every fatal exit path
|
||||
(2026-08-09, branch `db-close-on-fatal`, closes #4): `fatalf` and
|
||||
its `os.Exit(1)` are gone, so the deferred `db.Close()` — and with
|
||||
it the SQLite WAL checkpoint — now actually runs when a subcommand
|
||||
fails; `runScan`, `runReport`, `runTrees`, `loadRecords` and
|
||||
`resolveRoots` return errors instead. The single exit point is `run`
|
||||
in `main.go`: it maps a `fatalError` (anything a subcommand
|
||||
returned) to exit 1 and cobra's own argument and flag errors to exit
|
||||
2, which keeps a runtime failure from being reported as a usage
|
||||
error or printing the usage text. New `main_test.go` drives the CLI
|
||||
in-process and asserts the exit codes from README §Error handling
|
||||
plus the stdout/stderr split, including that a fatal error raised
|
||||
after the database is open leaves no `-wal`/`-shm` sidecar behind
|
||||
for `scan`, `report` or `trees`
|
||||
- update golangci-lint to v2.12.2 with the canonical config
|
||||
(2026-08-09, branch `golangci-v2.12.2`, merged as `38a01bd`,
|
||||
closes #3): bumped the pinned linter in the `Dockerfile` lint
|
||||
stage and `script/bootstrap` from v2.12.1 to v2.12.2, and replaced
|
||||
`.golangci.yml` with the canonical file — the linter settings
|
||||
(`lll`, `funlen`, `cyclop`, `dupl` thresholds) now live under
|
||||
`linters.settings` per the v2 schema, so they are actually applied; no new
|
||||
lint findings surfaced
|
||||
- convert Makefile targets to scripts-to-rule-them-all `script/` entrypoints
|
||||
like the other managed repos (2026-07-26, commit `3abeacf`, closes #1): all 12
|
||||
`script/` entrypoints exist (`bootstrap`, `setup`, `projectname`, `test`,
|
||||
`lint`, `fmt`, `fmt-check`, `check`, `docker`, `cibuild`, `precommit`,
|
||||
`install-precommit`) and every Makefile target is now a thin shim over them,
|
||||
matching the other managed repos
|
||||
`linters.settings` per the v2 schema, so they are actually
|
||||
applied; no new lint findings surfaced
|
||||
- convert Makefile targets to scripts-to-rule-them-all `script/`
|
||||
entrypoints like the other managed repos (2026-07-26, commit
|
||||
`3abeacf`, closes #1): all 12 `script/` entrypoints exist
|
||||
(`bootstrap`, `setup`, `projectname`, `test`, `lint`, `fmt`,
|
||||
`fmt-check`, `check`, `docker`, `cibuild`, `precommit`,
|
||||
`install-precommit`) and every Makefile target is now a thin shim
|
||||
over them, matching the other managed repos
|
||||
- make the binary the default Make target (2026-07-24, branch
|
||||
`make-default-target`): plain `make` now builds `sfdupes` (previously it ran
|
||||
`check` plus `build`); `make build` remains as an alias
|
||||
- scan-wide phases, concurrent operands, batched updates (2026-07-24, branch
|
||||
`scan-wide-phases`): all operands seed the shared walk pool and every pass
|
||||
runs once over the whole scan, so totals and ETAs are scan-global; the
|
||||
per-operand walk/hash/update cycles and their stderr announcements are gone;
|
||||
the update pass commits in batched transactions — the filesystem is
|
||||
authoritative and the database an eventually-consistent reflection, so
|
||||
scan-level atomicity is not required
|
||||
`make-default-target`): plain `make` now builds `sfdupes`
|
||||
(previously it ran `check` plus `build`); `make build` remains as
|
||||
an alias
|
||||
- scan-wide phases, concurrent operands, batched updates (2026-07-24,
|
||||
branch `scan-wide-phases`): all operands seed the shared walk pool
|
||||
and every pass runs once over the whole scan, so totals and ETAs
|
||||
are scan-global; the per-operand walk/hash/update cycles and their
|
||||
stderr announcements are gone; the update pass commits in batched
|
||||
transactions — the filesystem is authoritative and the database an
|
||||
eventually-consistent reflection, so scan-level atomicity is not
|
||||
required
|
||||
- split the stat pass back out of the walk (2026-07-24, branch
|
||||
`parallel-phases`): phases are strictly sequential again — walk, stat, hash,
|
||||
update per operand — with parallelism only inside each phase; the walk
|
||||
enumerates paths with per-directory workers and the stat pass lstats them with
|
||||
per-file workers, restoring the exact total/ETA stat bar
|
||||
- announce each operand on stderr before its passes (2026-07-24, branch
|
||||
`scan-operand-progress`): with per-operand walk/hash/update cycles, a
|
||||
multi-operand run (e.g. `scan /srv/*`) showed pass totals that looked like the
|
||||
whole run's — an operator watching operand 3 of 14 hash 300k files concluded
|
||||
20M files were being skipped
|
||||
- parallel walk (2026-07-24, branch `parallel-walk`): the walk pass was a single
|
||||
goroutine and took hours at ~20M files on a busy pool (observed: 22M files in
|
||||
4h on a ZFS server); it is now a per-directory worker-pool traversal that
|
||||
records size/mtime during the walk (folding away the separate stat pass,
|
||||
halving metadata I/O), and each `PATH` operand commits in its own transaction
|
||||
so an interrupted scan keeps completed operands
|
||||
`parallel-phases`): phases are strictly sequential again — walk,
|
||||
stat, hash, update per operand — with parallelism only inside each
|
||||
phase; the walk enumerates paths with per-directory workers and the
|
||||
stat pass lstats them with per-file workers, restoring the exact
|
||||
total/ETA stat bar
|
||||
- announce each operand on stderr before its passes (2026-07-24,
|
||||
branch `scan-operand-progress`): with per-operand walk/hash/update
|
||||
cycles, a multi-operand run (e.g. `scan /srv/*`) showed pass totals
|
||||
that looked like the whole run's — an operator watching operand 3 of
|
||||
14 hash 300k files concluded 20M files were being skipped
|
||||
- parallel walk (2026-07-24, branch `parallel-walk`): the walk pass
|
||||
was a single goroutine and took hours at ~20M files on a busy pool
|
||||
(observed: 22M files in 4h on a ZFS server); it is now a
|
||||
per-directory worker-pool traversal that records size/mtime during
|
||||
the walk (folding away the separate stat pass, halving metadata
|
||||
I/O), and each `PATH` operand commits in its own transaction so an
|
||||
interrupted scan keeps completed operands
|
||||
|
||||
- persistent scan database (2026-07-24, branch `persistent-database`): `scan`
|
||||
now maintains a SQLite database (`modernc.org/sqlite`, pure Go, cgo stays
|
||||
disabled) keyed by absolute path that survives between runs — a rescan hashes
|
||||
only new or changed files (by mtime/size), deletes records for files vanished
|
||||
from under the scanned operands, and leaves records outside them untouched, so
|
||||
`scan` can be cronned daily; `report` and `trees` read the database (no
|
||||
positional arguments) instead of a scan stream. Database at
|
||||
`/var/lib/sfdupes/db.sqlite`, overridable via `SFDUPES_DATABASE`; WAL
|
||||
journaling plus a single-transaction update keep a report run during a scan
|
||||
safe
|
||||
- add the `origin` remote (`git@git.eeqj.de:sneak/sfdupes.git`), tag `v0.0.1`,
|
||||
and push `main` plus tags (2026-07-23)
|
||||
- persistent scan database (2026-07-24, branch `persistent-database`):
|
||||
`scan` now maintains a SQLite database (`modernc.org/sqlite`, pure
|
||||
Go, cgo stays disabled) keyed by absolute path that survives between
|
||||
runs — a rescan hashes only new or changed files (by mtime/size),
|
||||
deletes records for files vanished from under the scanned operands,
|
||||
and leaves records outside them untouched, so `scan` can be cronned
|
||||
daily; `report` and `trees` read the database (no positional
|
||||
arguments) instead of a scan stream. Database at
|
||||
`/var/lib/sfdupes/db.sqlite`, overridable via `SFDUPES_DATABASE`;
|
||||
WAL journaling plus a single-transaction update keep a report run
|
||||
during a scan safe
|
||||
- add the `origin` remote (`git@git.eeqj.de:sneak/sfdupes.git`), tag
|
||||
`v0.0.1`, and push `main` plus tags (2026-07-23)
|
||||
- `scan` CLI rework (2026-07-23, branch `scan-required-paths`): required
|
||||
`PATH...` operands via cobra flags replacing the `/srv` `-root` default; new
|
||||
`-x`/`--one-file-system` flag (GNU convention) to stop at filesystem
|
||||
boundaries, which are crossed by default
|
||||
`PATH...` operands via cobra flags replacing the `/srv` `-root`
|
||||
default; new `-x`/`--one-file-system` flag (GNU convention) to stop
|
||||
at filesystem boundaries, which are crossed by default
|
||||
- bring the repo into full policy compliance (2026-07-23, branch
|
||||
`repo-policy-compliance`; checklist below)
|
||||
- `git init` with README-only first commit; code baseline committed on `main`
|
||||
(2026-07-22)
|
||||
- `git init` with README-only first commit; code baseline committed on
|
||||
`main` (2026-07-22)
|
||||
- implement `scan`, `report`, and `trees` subcommands (pre-git history)
|
||||
|
||||
# Future Steps
|
||||
|
||||
- possible later features (explicitly out of scope per README): full-content
|
||||
verification of candidates, removal-script helpers
|
||||
- possible later features (explicitly out of scope per README):
|
||||
full-content verification of candidates, removal-script helpers
|
||||
|
||||
# Repo Policy Compliance
|
||||
|
||||
Audited 2026-07-22 against `REPO_POLICIES.md` (2026-07-06), the existing repo
|
||||
checklist, and the Go styleguide. Code is already gofmt-clean, so no standalone
|
||||
formatting commit is needed.
|
||||
Audited 2026-07-22 against `REPO_POLICIES.md` (2026-07-06), the existing
|
||||
repo checklist, and the Go styleguide. Code is already gofmt-clean, so no
|
||||
standalone formatting commit is needed.
|
||||
|
||||
- [x] `.gitignore` missing — the compiled `sfdupes` binary and `files.dat` sit
|
||||
untracked in the tree; needs OS/editor/Go artifacts plus secrets patterns
|
||||
- [x] `.gitignore` missing — the compiled `sfdupes` binary and
|
||||
`files.dat` sit untracked in the tree; needs OS/editor/Go
|
||||
artifacts plus secrets patterns
|
||||
- [x] `.editorconfig` missing
|
||||
- [x] `LICENSE` missing and README has no License section (MIT assumed from
|
||||
house convention — user to confirm)
|
||||
- [x] `LICENSE` missing and README has no License section (MIT assumed
|
||||
from house convention — user to confirm)
|
||||
- [x] `REPO_POLICIES.md` missing from repo root
|
||||
- [x] `.golangci.yml` missing (install canonical copy); code must then pass
|
||||
`make lint` (150 findings fixed; `make lint` is clean)
|
||||
- [x] `Makefile` lacks required targets `test`, `lint`, `fmt`, `fmt-check`,
|
||||
`docker`, `hooks`; `check` currently depends on `build`, which writes the
|
||||
binary (`make check` must not modify files)
|
||||
- [x] no tests — `go test ./...` has nothing to run; policy requires real tests
|
||||
with a 30-second timeout and the conditional `-v` rerun pattern (suite
|
||||
covers parsing, grouping, digests, suppression, hashing, and the scan
|
||||
pipeline; 64% coverage)
|
||||
- [x] `Dockerfile` missing — Go multistage with hash-pinned images: fail-fast
|
||||
lint stage, build stage running `make check`
|
||||
- [x] `.golangci.yml` missing (install canonical copy); code must then
|
||||
pass `make lint` (150 findings fixed; `make lint` is clean)
|
||||
- [x] `Makefile` lacks required targets `test`, `lint`, `fmt`,
|
||||
`fmt-check`, `docker`, `hooks`; `check` currently depends on
|
||||
`build`, which writes the binary (`make check` must not modify
|
||||
files)
|
||||
- [x] no tests — `go test ./...` has nothing to run; policy requires
|
||||
real tests with a 30-second timeout and the conditional `-v`
|
||||
rerun pattern (suite covers parsing, grouping, digests,
|
||||
suppression, hashing, and the scan pipeline; 64% coverage)
|
||||
- [x] `Dockerfile` missing — Go multistage with hash-pinned images:
|
||||
fail-fast lint stage, build stage running `make check`
|
||||
- [x] `.dockerignore` missing
|
||||
- [x] `.gitea/workflows/check.yml` missing (`docker build .` on push, checkout
|
||||
action pinned by commit SHA)
|
||||
- [x] `.gitea/workflows/check.yml` missing (`docker build .` on push,
|
||||
checkout action pinned by commit SHA)
|
||||
- [x] README lacks required sections: Description first line
|
||||
(name/purpose/category/license/author), Getting Started, Rationale, TODO,
|
||||
License, Author
|
||||
- [x] README non-goal "no git repository setup and no CI" is stale now that the
|
||||
repo is under git with CI
|
||||
- [x] pre-commit hook not installed (`make hooks` once the target exists)
|
||||
(name/purpose/category/license/author), Getting Started,
|
||||
Rationale, TODO, License, Author
|
||||
- [x] README non-goal "no git repository setup and no CI" is stale now
|
||||
that the repo is under git with CI
|
||||
- [x] pre-commit hook not installed (`make hooks` once the target
|
||||
exists)
|
||||
|
||||
Accepted divergences (no action):
|
||||
|
||||
- flat single-package layout with `.go` files in the repo root — fine for a
|
||||
small single-binary tool per the Go styleguide; the tracker audit agrees
|
||||
- `go test` runs without `-race` — the repo mandates `CGO_ENABLED=0` (pure-Go
|
||||
builds) and the race detector requires cgo
|
||||
- flat single-package layout with `.go` files in the repo root — fine
|
||||
for a small single-binary tool per the Go styleguide; the tracker
|
||||
audit agrees
|
||||
- `go test` runs without `-race` — the repo mandates `CGO_ENABLED=0`
|
||||
(pure-Go builds) and the race detector requires cgo
|
||||
|
||||
+1
-1
@@ -456,7 +456,7 @@ func TestHashWorkerDropsQueuedRuns(t *testing.T) {
|
||||
go func() {
|
||||
defer close(done)
|
||||
|
||||
hashWorker(cancelledContext(t), jobs, results)
|
||||
hashWorker(cancelledContext(t), jobs, results, hashSignature)
|
||||
}()
|
||||
|
||||
awaitReturn(t, done, "hashWorker")
|
||||
|
||||
@@ -40,18 +40,20 @@ CREATE TABLE files (
|
||||
size INTEGER NOT NULL,
|
||||
mtime INTEGER NOT NULL,
|
||||
head TEXT NOT NULL,
|
||||
tail TEXT NOT NULL
|
||||
tail TEXT NOT NULL,
|
||||
content TEXT NOT NULL
|
||||
) WITHOUT ROWID
|
||||
`
|
||||
|
||||
// upsertSQL inserts one file record, replacing any existing record for
|
||||
// the same path.
|
||||
const upsertSQL = `
|
||||
INSERT INTO files (path, size, mtime, head, tail)
|
||||
VALUES (?, ?, ?, ?, ?)
|
||||
INSERT INTO files (path, size, mtime, head, tail, content)
|
||||
VALUES (?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT (path) DO UPDATE SET
|
||||
size = excluded.size, mtime = excluded.mtime,
|
||||
head = excluded.head, tail = excluded.tail
|
||||
head = excluded.head, tail = excluded.tail,
|
||||
content = excluded.content
|
||||
`
|
||||
|
||||
// errNoDatabase reports a missing database file for report/trees.
|
||||
@@ -205,7 +207,7 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) {
|
||||
// loadFileRows reads every record from the files table.
|
||||
func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
||||
rows, err := db.QueryContext(ctx,
|
||||
"SELECT path, size, mtime, head, tail FROM files")
|
||||
"SELECT path, size, mtime, head, tail, content FROM files")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
@@ -220,7 +222,8 @@ func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
||||
r scanRec
|
||||
)
|
||||
|
||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail)
|
||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
|
||||
&r.content)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read record: %w", err)
|
||||
}
|
||||
@@ -275,6 +278,62 @@ func loadFileMeta(ctx context.Context, db *sql.DB,
|
||||
return nil
|
||||
}
|
||||
|
||||
// contentCandidatesSQL selects every record of at least headTailMin
|
||||
// bytes whose size, head, and tail equal another record's, in each
|
||||
// group (the records sharing a size, head, and tail) where at least one
|
||||
// record has no content hash, with whether each record has one. SQLite
|
||||
// does the grouping, so no other record's hashes are loaded into
|
||||
// memory; the rows come ordered by size, head, and tail, so each
|
||||
// group's rows arrive together.
|
||||
const contentCandidatesSQL = `
|
||||
SELECT f.path, f.size, f.mtime, f.head, f.tail, f.content <> ''
|
||||
FROM files AS f
|
||||
JOIN (
|
||||
SELECT size, head, tail
|
||||
FROM files
|
||||
WHERE size >= ? AND head <> ''
|
||||
GROUP BY size, head, tail
|
||||
HAVING COUNT(*) > 1 AND SUM(content = '') > 0
|
||||
) AS g USING (size, head, tail)
|
||||
ORDER BY size, head, tail
|
||||
`
|
||||
|
||||
// loadContentCandidates streams the rows of contentCandidatesSQL to fn:
|
||||
// each record, without its content hash, and whether it has one.
|
||||
func loadContentCandidates(ctx context.Context, db *sql.DB,
|
||||
fn func(r scanRec, hashed bool),
|
||||
) error {
|
||||
rows, err := db.QueryContext(ctx, contentCandidatesSQL, headTailMin)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
for rows.Next() {
|
||||
var (
|
||||
path []byte
|
||||
r scanRec
|
||||
hashed int64
|
||||
)
|
||||
|
||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, &hashed)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read record: %w", err)
|
||||
}
|
||||
|
||||
r.path = string(path)
|
||||
fn(r, hashed != 0)
|
||||
}
|
||||
|
||||
err = rows.Err()
|
||||
if err != nil {
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// updateBatchSize is the number of record changes committed per
|
||||
// transaction during the update pass. The filesystem is authoritative
|
||||
// and the database an eventually-consistent reflection of it, so
|
||||
@@ -348,7 +407,7 @@ func execUpserts(ctx context.Context, tx *sql.Tx, upserts []scanRec,
|
||||
|
||||
for _, r := range upserts {
|
||||
_, err = st.ExecContext(ctx,
|
||||
[]byte(r.path), r.size, r.mtime, r.head, r.tail)
|
||||
[]byte(r.path), r.size, r.mtime, r.head, r.tail, r.content)
|
||||
if err != nil {
|
||||
return fmt.Errorf("upsert %s: %w", r.path, err)
|
||||
}
|
||||
|
||||
+10
-4
@@ -136,10 +136,14 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
||||
db := openTestDB(t)
|
||||
|
||||
// Paths may contain tabs and newlines; the database must store
|
||||
// them byte-exactly.
|
||||
// them byte-exactly. Every hash, content included, comes back as
|
||||
// written.
|
||||
recs := []scanRec{
|
||||
{size: 2, mtime: 20, head: "h2", tail: "t2", path: "/a/tab\tnew\nline"},
|
||||
{size: 1, mtime: 10, head: "h1", tail: "t1", path: "/a/x"},
|
||||
{
|
||||
size: 2, mtime: 20, head: "h2", tail: "t2", content: "c2",
|
||||
path: "/a/tab\tnew\nline",
|
||||
},
|
||||
{size: 1, mtime: 10, head: "h1", tail: "t1", content: "c1", path: "/a/x"},
|
||||
}
|
||||
|
||||
err := applyChanges(t.Context(), db, recs, nil,
|
||||
@@ -163,7 +167,9 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
||||
|
||||
// An upsert for an existing path updates in place; a delete
|
||||
// removes exactly its path.
|
||||
upd := scanRec{size: 3, mtime: 30, head: "h3", tail: "t3", path: "/a/x"}
|
||||
upd := scanRec{
|
||||
size: 3, mtime: 30, head: "h3", tail: "t3", content: "c3", path: "/a/x",
|
||||
}
|
||||
|
||||
err = applyChanges(t.Context(), db, []scanRec{upd},
|
||||
[]string{"/a/tab\tnew\nline"}, newProgress("update", 2))
|
||||
|
||||
@@ -1,10 +1,14 @@
|
||||
// Command sfdupes quickly identifies candidate duplicate files across
|
||||
// very large filesystems without reading full file contents. Files are
|
||||
// considered duplicates when they have identical size, identical SHA-256
|
||||
// of their first 1024 bytes, and identical SHA-256 of their last 1024
|
||||
// bytes. scan maintains a persistent SQLite database of file signatures
|
||||
// (SFDUPES_DATABASE, default /var/lib/sfdupes/db.sqlite) that the
|
||||
// reporting subcommands read.
|
||||
// very large filesystems without reading every byte of every file.
|
||||
// Files are considered duplicates when their sizes are equal and they
|
||||
// agree on a short ladder of SHA-256 hashes. A file under 10 MiB is
|
||||
// hashed in full. A larger file is compared on the hashes of its first
|
||||
// and last 64 KiB, and only when those match another file's is its
|
||||
// content hash computed and compared: of the whole file when it is
|
||||
// under 50 MiB, or of gigabyte-spaced 1 MiB samples when it is 50 MiB
|
||||
// or larger. scan maintains a persistent SQLite database of file
|
||||
// signatures (SFDUPES_DATABASE, default /var/lib/sfdupes/db.sqlite)
|
||||
// that the reporting subcommands read.
|
||||
//
|
||||
// Usage:
|
||||
//
|
||||
@@ -97,7 +101,7 @@ func run(args []string, stderr io.Writer) int {
|
||||
func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||
root := &cobra.Command{
|
||||
Use: "sfdupes",
|
||||
Short: "Find candidate duplicate files by size and head/tail SHA-256",
|
||||
Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
|
||||
Version: Version,
|
||||
Args: cobra.NoArgs,
|
||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||
@@ -127,7 +131,7 @@ func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||
}),
|
||||
}
|
||||
scanCmd.Flags().IntVar(&scanWorkers, "workers", runtime.NumCPU(),
|
||||
"concurrent workers for the walk and hash phases")
|
||||
"concurrent workers for the walk, hash, and content phases")
|
||||
scanCmd.Flags().BoolVarP(&scanOneFS, "one-file-system", "x", false,
|
||||
"do not cross filesystem boundaries")
|
||||
|
||||
|
||||
@@ -1,5 +0,0 @@
|
||||
{
|
||||
"devDependencies": {
|
||||
"prettier": "3.8.1"
|
||||
}
|
||||
}
|
||||
@@ -17,13 +17,14 @@ const ioBufSize = 1 << 20
|
||||
const minGroupSize = 2
|
||||
|
||||
// scanRec is one file record from the database. The signature (size,
|
||||
// head, tail) is the duplicate key; mtime is informational only and
|
||||
// used by scan for change detection.
|
||||
// head, tail, content) is the duplicate key; mtime is informational
|
||||
// only and used by scan for change detection.
|
||||
type scanRec struct {
|
||||
size int64
|
||||
mtime int64
|
||||
head string
|
||||
tail string
|
||||
content string
|
||||
path string
|
||||
}
|
||||
|
||||
@@ -52,8 +53,9 @@ func loadRecords(ctx context.Context) ([]scanRec, error) {
|
||||
}
|
||||
|
||||
// dupeGroup is one set of candidate-duplicate files: identical size,
|
||||
// head hash, and tail hash. paths is sorted lexicographically; the
|
||||
// first entry is the group's "first", the rest are dupes.
|
||||
// head hash, tail hash, and content hash. paths is sorted
|
||||
// lexicographically; the first entry is the group's "first", the rest
|
||||
// are dupes.
|
||||
type dupeGroup struct {
|
||||
size int64
|
||||
paths []string
|
||||
@@ -115,14 +117,15 @@ func collectDupeGroups(recs []scanRec) []dupeGroup {
|
||||
groups := make(map[fileSig][]string)
|
||||
|
||||
for _, r := range recs {
|
||||
// A record without hashes (its size was unique when last
|
||||
// scanned) has unknown content and is never reported as a
|
||||
// duplicate.
|
||||
if r.head == "" {
|
||||
// A record without a content hash has unknown content and is
|
||||
// never reported as a duplicate (README "Database").
|
||||
if r.content == "" {
|
||||
continue
|
||||
}
|
||||
|
||||
k := fileSig{size: r.size, head: r.head, tail: r.tail}
|
||||
k := fileSig{
|
||||
size: r.size, head: r.head, tail: r.tail, content: r.content,
|
||||
}
|
||||
groups[k] = append(groups[k], r.path)
|
||||
}
|
||||
|
||||
|
||||
+43
-17
@@ -9,15 +9,15 @@ func TestCollectDupeGroups(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
recs := []scanRec{
|
||||
{size: 100, head: "h", tail: "t", path: "/z/b"},
|
||||
{size: 100, head: "h", tail: "t", path: "/z/a"},
|
||||
{size: 100, head: "h", tail: "t", path: "/z/c"},
|
||||
{size: 4000, head: "H", tail: "T", path: "/big/2"},
|
||||
{size: 4000, head: "H", tail: "T", path: "/big/1"},
|
||||
{size: 100, head: "h", tail: "t", content: "c", path: "/z/b"},
|
||||
{size: 100, head: "h", tail: "t", content: "c", path: "/z/a"},
|
||||
{size: 100, head: "h", tail: "t", content: "c", path: "/z/c"},
|
||||
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/2"},
|
||||
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/1"},
|
||||
// Same size as the /z group but a different head hash.
|
||||
{size: 100, head: "other", tail: "t", path: "/z/d"},
|
||||
{size: 100, head: "other", tail: "t", content: "c", path: "/z/d"},
|
||||
// A singleton signature must not form a group.
|
||||
{size: 7, head: "u", tail: "u", path: "/lonely"},
|
||||
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
@@ -38,14 +38,40 @@ func TestCollectDupeGroups(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroupsContentSeparates(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// Same size, head, and tail, but different content hashes: the final
|
||||
// rung keeps them apart, so no group forms. Matching content groups.
|
||||
// Records without a content hash never group, not even with each
|
||||
// other.
|
||||
recs := []scanRec{
|
||||
{size: 100, head: "h", tail: "t", content: "c1", path: "/a"},
|
||||
{size: 100, head: "h", tail: "t", content: "c2", path: "/b"},
|
||||
{size: 100, head: "h", tail: "t", content: "c1", path: "/c"},
|
||||
{size: 100, head: "h", tail: "t", path: "/d"},
|
||||
{size: 100, head: "h", tail: "t", path: "/e"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
||||
len(groups))
|
||||
}
|
||||
|
||||
if !slices.Equal(groups[0].paths, []string{"/a", "/c"}) {
|
||||
t.Errorf("group paths = %q, want /a /c", groups[0].paths)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// mtime is informational only; records differing only in mtime
|
||||
// still group together.
|
||||
recs := []scanRec{
|
||||
{size: 9, mtime: 100, head: "h", tail: "t", path: "/m/1"},
|
||||
{size: 9, mtime: 200, head: "h", tail: "t", path: "/m/2"},
|
||||
{size: 9, mtime: 100, head: "h", tail: "t", content: "c", path: "/m/1"},
|
||||
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
@@ -58,10 +84,10 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
recs := []scanRec{
|
||||
{size: 50, head: "b", tail: "b", path: "/beta/2"},
|
||||
{size: 50, head: "b", tail: "b", path: "/beta/1"},
|
||||
{size: 50, head: "a", tail: "a", path: "/alpha/2"},
|
||||
{size: 50, head: "a", tail: "a", path: "/alpha/1"},
|
||||
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"},
|
||||
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"},
|
||||
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"},
|
||||
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
@@ -80,10 +106,10 @@ func TestCollectDupeGroupsDeterministic(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
recs := []scanRec{
|
||||
{size: 1, head: "a", tail: "a", path: "/p/1"},
|
||||
{size: 1, head: "a", tail: "a", path: "/p/2"},
|
||||
{size: 2, head: "b", tail: "b", path: "/q/1"},
|
||||
{size: 2, head: "b", tail: "b", path: "/q/2"},
|
||||
{size: 1, head: "a", tail: "a", content: "a", path: "/p/1"},
|
||||
{size: 1, head: "a", tail: "a", content: "a", path: "/p/2"},
|
||||
{size: 2, head: "b", tail: "b", content: "b", path: "/q/1"},
|
||||
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
||||
}
|
||||
|
||||
forward := collectDupeGroups(recs)
|
||||
|
||||
@@ -6,7 +6,9 @@ import (
|
||||
"crypto/sha256"
|
||||
"database/sql"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"io/fs"
|
||||
"os"
|
||||
"path/filepath"
|
||||
@@ -16,8 +18,40 @@ import (
|
||||
"syscall"
|
||||
)
|
||||
|
||||
// chunk is the number of bytes hashed from each end of a file.
|
||||
const chunk = 1024
|
||||
// The duplicate ladder (see hashSignature and README "Duplicate
|
||||
// detection"). A same-size candidate below headTailMin is hashed in
|
||||
// full and compared directly; a larger one is separated first by the
|
||||
// hashes of its end windows, then by a content hash that is exact below
|
||||
// wholeFileMax and deliberately sampled at or above it. The hash phase
|
||||
// reads only the end windows of a larger file; the content phase reads
|
||||
// it for its content hash only once its size, head, and tail match
|
||||
// another file's.
|
||||
|
||||
// headTailMin is the size threshold for the end-window gate. A file
|
||||
// smaller than this is hashed in full directly, with no separate head
|
||||
// and tail step: its head, tail, and content all carry the whole-file
|
||||
// hash. A file this size or larger is separated first by its end
|
||||
// windows.
|
||||
const headTailMin = 10 * 1024 * 1024
|
||||
|
||||
// headTailWindow is the number of bytes hashed from each end of a file
|
||||
// at or above headTailMin (the head and tail rungs). Because
|
||||
// headTailMin is far larger than two windows, the head and tail windows
|
||||
// never overlap.
|
||||
const headTailWindow = 64 * 1024
|
||||
|
||||
// wholeFileMax is the size boundary between the two content rungs: a
|
||||
// file strictly smaller than this is content-hashed in full; a file
|
||||
// this size or larger is content-hashed by sampling.
|
||||
const wholeFileMax = 50 * 1024 * 1024
|
||||
|
||||
// sampleStride is the spacing between content samples for large files:
|
||||
// one window is read at each gigabyte-aligned offset (0, 1 GiB, ...).
|
||||
const sampleStride = 1024 * 1024 * 1024
|
||||
|
||||
// sampleWindow is the number of bytes read at each large-file sample
|
||||
// offset, truncated at end of file.
|
||||
const sampleWindow = 1024 * 1024
|
||||
|
||||
// workQueueDepth bounds the job and result channels feeding the walk
|
||||
// and hash worker pools.
|
||||
@@ -44,16 +78,17 @@ type fileMeta struct {
|
||||
hashed bool
|
||||
}
|
||||
|
||||
// runScan implements the scan subcommand: three sequential phases —
|
||||
// walk (which stats each file as it is discovered), hash, update —
|
||||
// that synchronize the persistent database with the filesystem state
|
||||
// under the PATH operands. Only files whose size at least one other
|
||||
// file shares are ever hashed: a size-unique file cannot be a
|
||||
// duplicate. Flag parsing and the at-least-one-operand check are done
|
||||
// by cobra. Errors are returned rather than exiting, so that the
|
||||
// deferred close — which checkpoints the SQLite WAL — always runs.
|
||||
// Cancelling ctx unwinds the worker pools and aborts the scan with the
|
||||
// context's error.
|
||||
// runScan implements the scan subcommand: four sequential phases —
|
||||
// walk (which stats each file as it is discovered), hash, update,
|
||||
// content — that synchronize the persistent database with the
|
||||
// filesystem state under the PATH operands. Only files whose size at
|
||||
// least one other file shares are ever hashed: a size-unique file
|
||||
// cannot be a duplicate. A file of headTailMin or more gets its content
|
||||
// hash only when its size, head, and tail match another file's. Flag
|
||||
// parsing and the at-least-one-operand check are done by cobra. Errors
|
||||
// are returned rather than exiting, so that the deferred close — which
|
||||
// checkpoints the SQLite WAL — always runs. Cancelling ctx unwinds the
|
||||
// worker pools and aborts the scan with the context's error.
|
||||
func runScan(ctx context.Context, roots []string, workers int,
|
||||
oneFS bool,
|
||||
) error {
|
||||
@@ -166,13 +201,16 @@ type scanState struct {
|
||||
}
|
||||
|
||||
// syncScan synchronizes the database with the filesystem under roots
|
||||
// in three sequential phases: walk (enumerate and stat every file,
|
||||
// in four sequential phases: walk (enumerate and stat every file,
|
||||
// building a complete size census), hash (read only the new or
|
||||
// changed — or previously unhashed — files whose size at least one
|
||||
// other file shares, committing results in batches as they arrive),
|
||||
// and update (record the size-unique files without reading them, and
|
||||
// delete the records the scan no longer verifies). Records outside
|
||||
// the roots are never touched.
|
||||
// update (record the size-unique files without reading them, and
|
||||
// delete the records the scan no longer verifies), and content (fill
|
||||
// in the content hash of every record of headTailMin or more whose
|
||||
// size, head, and tail match another record's). Records outside the
|
||||
// roots are never touched, except that the content phase fills in
|
||||
// their content hash.
|
||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||
workers int, oneFS bool,
|
||||
) (scanStats, error) {
|
||||
@@ -207,7 +245,12 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||
return s.st, err
|
||||
}
|
||||
|
||||
return s.st, s.updatePhase(ctx)
|
||||
err = s.updatePhase(ctx)
|
||||
if err != nil {
|
||||
return s.st, err
|
||||
}
|
||||
|
||||
return s.st, s.contentPhase(ctx, workers)
|
||||
}
|
||||
|
||||
// loadIndex indexes the database records under the scan roots for
|
||||
@@ -241,7 +284,8 @@ func (s *scanState) loadIndex(ctx context.Context, roots []string) error {
|
||||
|
||||
// walkPhase drains the walk, appending every walked file's size to
|
||||
// the census and resolving what it can immediately: an unchanged file
|
||||
// whose record already has hashes needs nothing further. It returns
|
||||
// whose record already has hashes needs nothing from the hash phase
|
||||
// (the content phase may still fill in its content hash). It returns
|
||||
// the new-or-changed files and the unchanged files whose records lack
|
||||
// hashes; both remain candidates until the census decides whether
|
||||
// their sizes are shared.
|
||||
@@ -380,27 +424,40 @@ func sameInode(a, b fileRec) bool {
|
||||
return (a.dev != 0 || a.ino != 0) && a.dev == b.dev && a.ino == b.ino
|
||||
}
|
||||
|
||||
// hashPhase hashes every queued file with the worker pool — one read
|
||||
// per inode run, in inode order — committing completed records to the
|
||||
// database in batches as results arrive, so a long scan persists its
|
||||
// progress as it goes (an interrupted scan resumes cheaply: the next
|
||||
// run skips everything already recorded). The total counts actual
|
||||
// reads, so the bar shows a real ETA. A run that fails to hash is
|
||||
// warned about and skipped; stale records for its paths, if any, are
|
||||
// deleted by the update phase.
|
||||
// hashPhase hashes every queued file with hashSignature — the head and
|
||||
// tail of a file of headTailMin or more, the whole file below that —
|
||||
// committing completed records to the database in batches as results
|
||||
// arrive, so a long scan persists its progress as it goes (an
|
||||
// interrupted scan resumes cheaply: the next run skips everything
|
||||
// already recorded). A run that fails to hash is warned about and
|
||||
// skipped; stale records for its paths, if any, are deleted by the
|
||||
// update phase.
|
||||
func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
||||
runs := hashRuns(s.toHash)
|
||||
s.toHash = nil
|
||||
|
||||
return s.readRuns(ctx, workers, "hash", runs, hashSignature, s.recordRun)
|
||||
}
|
||||
|
||||
// readRuns reads runs with the worker pool, one read per inode run, in
|
||||
// the order given, under a progress display named label. The workers
|
||||
// compute each run's hashes with hash, and each result goes to record;
|
||||
// a run that fails to read is warned about and counted as skipped
|
||||
// instead. The total counts actual reads, so the bar shows a real ETA.
|
||||
//
|
||||
// Returning early — a failed database write, or a cancelled scan — must
|
||||
// not strand the pool: the feeder would park forever on a full jobs
|
||||
// channel and every worker on a full results channel. The deferred stop
|
||||
// is what prevents that.
|
||||
func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
||||
runs := hashRuns(s.toHash)
|
||||
s.toHash = nil
|
||||
|
||||
pool := startHashPool(ctx, runs, workers)
|
||||
func (s *scanState) readRuns(ctx context.Context, workers int,
|
||||
label string, runs [][]fileRec,
|
||||
hash func(path string, size int64) (string, string, string, error),
|
||||
record func(ctx context.Context, r hashResult) error,
|
||||
) error {
|
||||
pool := startHashPool(ctx, runs, workers, hash)
|
||||
defer pool.stop()
|
||||
|
||||
prog := newProgress("hash", int64(len(runs)))
|
||||
prog := newProgress(label, int64(len(runs)))
|
||||
defer prog.finish()
|
||||
|
||||
for range runs {
|
||||
@@ -417,12 +474,12 @@ func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
||||
if r.err != nil {
|
||||
s.st.skipped += len(r.run)
|
||||
|
||||
prog.warnf("hash %s: %v", r.run[0].path, r.err)
|
||||
prog.warnf("%s %s: %v", label, r.run[0].path, r.err)
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
err := s.recordRun(ctx, r)
|
||||
err := record(ctx, r)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -443,10 +500,17 @@ func (s *scanState) recordRun(ctx context.Context, r hashResult) error {
|
||||
mtime: rec.mtime,
|
||||
head: r.head,
|
||||
tail: r.tail,
|
||||
content: r.content,
|
||||
path: rec.path,
|
||||
})
|
||||
}
|
||||
|
||||
return s.commitFullBatch(ctx)
|
||||
}
|
||||
|
||||
// commitFullBatch commits the running batch once it holds
|
||||
// updateBatchSize records.
|
||||
func (s *scanState) commitFullBatch(ctx context.Context) error {
|
||||
if len(s.batch) < updateBatchSize {
|
||||
return nil
|
||||
}
|
||||
@@ -504,6 +568,147 @@ func (s *scanState) updatePhase(ctx context.Context) error {
|
||||
return applyChanges(ctx, s.db, nil, deletes, prog)
|
||||
}
|
||||
|
||||
// contentPhase fills in the content hash of every record of headTailMin
|
||||
// or more that lacks one and whose size, head, and tail equal another
|
||||
// record's, anywhere in the database: records from this scan and
|
||||
// records stored by earlier scans, inside or outside the roots. Only
|
||||
// such a file can still be a duplicate, so no other file of headTailMin
|
||||
// or more is read beyond its end windows. The files are read with the
|
||||
// hash phase's worker pool and their records written back in batches. A
|
||||
// failed read is warned about and counted as skipped; the record keeps
|
||||
// its empty content, so it is never grouped, and a later scan tries
|
||||
// again.
|
||||
func (s *scanState) contentPhase(ctx context.Context, workers int) error {
|
||||
toRead, recs, err := s.contentCandidates(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
err = s.readRuns(ctx, workers, "content", hashRuns(toRead),
|
||||
hashContentOnly, func(ctx context.Context, r hashResult) error {
|
||||
// Every path in the run keeps its record's head and tail
|
||||
// and gains the one content hash read for the run.
|
||||
for _, f := range r.run {
|
||||
rec := recs[f.path]
|
||||
rec.content = r.content
|
||||
s.batch = append(s.batch, rec)
|
||||
}
|
||||
|
||||
return s.commitFullBatch(ctx)
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
return applyChanges(ctx, s.db, s.batch, nil, nil)
|
||||
}
|
||||
|
||||
// contentCandidates returns the files the content phase reads, and
|
||||
// their records by path. Every record contentCandidatesSQL returns has
|
||||
// its file checked with lstat, whether or not it already has a content
|
||||
// hash: a file that is gone, is no longer a regular file, or has
|
||||
// changed by the walk's rule keeps its record as it is and does not
|
||||
// count as a match for the others, and any other lstat error is warned
|
||||
// about and counted as skipped, with the same result. If such a record
|
||||
// has no content hash, it stays out of duplicate groups; if it has one,
|
||||
// it is still reported until a scan covering its own tree updates or
|
||||
// removes it. The files of a group that pass and have no content hash
|
||||
// are read only if at least minGroupSize of the group's files pass, so
|
||||
// a group whose other members are all stale costs no reads. Only the
|
||||
// records to be read are kept.
|
||||
func (s *scanState) contentCandidates(
|
||||
ctx context.Context,
|
||||
) ([]fileRec, map[string]scanRec, error) {
|
||||
// The query and the checks take real time on a large database;
|
||||
// without a display the scan looks hung before the reads begin.
|
||||
prog := newProgress("content", -1)
|
||||
defer prog.finish()
|
||||
|
||||
var (
|
||||
toRead []fileRec
|
||||
first scanRec // the current group's first record
|
||||
passed int // the current group's files that passed the check
|
||||
unread []fileRec // those of them without a content hash
|
||||
)
|
||||
|
||||
recs := make(map[string]scanRec)
|
||||
|
||||
// endGroup queues the current group's files to read if at least
|
||||
// minGroupSize of its files passed, and drops their records if not.
|
||||
endGroup := func() {
|
||||
if passed >= minGroupSize {
|
||||
toRead = append(toRead, unread...)
|
||||
} else {
|
||||
for _, f := range unread {
|
||||
delete(recs, f.path)
|
||||
}
|
||||
}
|
||||
|
||||
passed, unread = 0, nil
|
||||
}
|
||||
|
||||
err := loadContentCandidates(ctx, s.db, func(r scanRec, hashed bool) {
|
||||
prog.increment()
|
||||
|
||||
if r.size != first.size || r.head != first.head || r.tail != first.tail {
|
||||
endGroup()
|
||||
|
||||
first = r
|
||||
}
|
||||
|
||||
f, ok, err := unchangedFile(r)
|
||||
if err != nil {
|
||||
s.st.skipped++
|
||||
|
||||
prog.warnf("content %s: %v", r.path, err)
|
||||
}
|
||||
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
|
||||
passed++
|
||||
|
||||
if !hashed {
|
||||
unread = append(unread, f)
|
||||
recs[r.path] = r
|
||||
}
|
||||
})
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
|
||||
endGroup()
|
||||
|
||||
return toRead, recs, nil
|
||||
}
|
||||
|
||||
// unchangedFile lstats the file r names and returns it for reading if
|
||||
// it is still the regular file r records: the same size, and an mtime
|
||||
// no newer than recorded (the walk's change rule). A file that is gone
|
||||
// or has changed reports false; any other lstat error is returned.
|
||||
func unchangedFile(r scanRec) (fileRec, bool, error) {
|
||||
fi, err := os.Lstat(r.path)
|
||||
if errors.Is(err, fs.ErrNotExist) {
|
||||
return fileRec{}, false, nil
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
return fileRec{}, false, err
|
||||
}
|
||||
|
||||
if !fi.Mode().IsRegular() || fi.Size() != r.size ||
|
||||
fi.ModTime().Unix() > r.mtime {
|
||||
return fileRec{}, false, nil
|
||||
}
|
||||
|
||||
dev, ino := inodeOfInfo(fi)
|
||||
|
||||
return fileRec{
|
||||
path: r.path, size: r.size, mtime: r.mtime, dev: dev, ino: ino,
|
||||
}, true, nil
|
||||
}
|
||||
|
||||
// underAnyRoot reports whether path is any of the roots or lies under
|
||||
// one of them.
|
||||
func underAnyRoot(path string, roots []string) bool {
|
||||
@@ -834,12 +1039,15 @@ func inodeOfInfo(fi fs.FileInfo) (uint64, uint64) {
|
||||
return statDev(st), st.Ino
|
||||
}
|
||||
|
||||
// hashResult carries one inode run's head/tail hashes (or the error
|
||||
// that prevented hashing it) from the hash workers to the hash phase.
|
||||
// hashResult carries the hashes computed for one inode run (or the
|
||||
// error that prevented computing them) from the pool's workers to the
|
||||
// phase that started the pool: head, tail, and content from
|
||||
// hashSignature, content alone from hashContentOnly.
|
||||
type hashResult struct {
|
||||
run []fileRec
|
||||
head string
|
||||
tail string
|
||||
content string
|
||||
err error
|
||||
}
|
||||
|
||||
@@ -856,10 +1064,10 @@ type hashPool struct {
|
||||
}
|
||||
|
||||
// startHashPool starts the feeder and the workers over runs. Workers
|
||||
// hash each run's first path (all paths in a run are hard links to the
|
||||
// same inode) and write one result per run.
|
||||
func startHashPool(ctx context.Context, runs [][]fileRec,
|
||||
workers int,
|
||||
// hash each run's first path with hash (all paths in a run are hard
|
||||
// links to the same inode) and write one result per run.
|
||||
func startHashPool(ctx context.Context, runs [][]fileRec, workers int,
|
||||
hash func(path string, size int64) (string, string, string, error),
|
||||
) *hashPool {
|
||||
ctx, cancel := context.WithCancel(ctx)
|
||||
|
||||
@@ -871,7 +1079,7 @@ func startHashPool(ctx context.Context, runs [][]fileRec,
|
||||
wg.Go(func() { feedHashJobs(ctx, runs, jobs) })
|
||||
|
||||
for range workers {
|
||||
wg.Go(func() { hashWorker(ctx, jobs, results) })
|
||||
wg.Go(func() { hashWorker(ctx, jobs, results, hash) })
|
||||
}
|
||||
|
||||
done := make(chan struct{})
|
||||
@@ -917,24 +1125,25 @@ func feedHashJobs(ctx context.Context, runs [][]fileRec,
|
||||
}
|
||||
}
|
||||
|
||||
// hashWorker hashes one inode run at a time until jobs is closed or the
|
||||
// scan is cancelled. A cancelled worker drops the runs still queued
|
||||
// instead of stopping its reads of jobs: the range must run out for the
|
||||
// pool to tear down, and reading a file nobody wants the hash of only
|
||||
// delays that.
|
||||
// hashWorker hashes one inode run at a time with hash until jobs is
|
||||
// closed or the scan is cancelled. A cancelled worker drops the runs
|
||||
// still queued instead of stopping its reads of jobs: the range must
|
||||
// run out for the pool to tear down, and reading a file nobody wants
|
||||
// the hash of only delays that.
|
||||
func hashWorker(ctx context.Context, jobs <-chan []fileRec,
|
||||
results chan<- hashResult,
|
||||
hash func(path string, size int64) (string, string, string, error),
|
||||
) {
|
||||
for run := range jobs {
|
||||
if ctx.Err() != nil {
|
||||
continue
|
||||
}
|
||||
|
||||
head, tail, err := hashHeadTail(run[0].path, run[0].size)
|
||||
head, tail, content, err := hash(run[0].path, run[0].size)
|
||||
|
||||
select {
|
||||
case results <- hashResult{
|
||||
run: run, head: head, tail: tail, err: err,
|
||||
run: run, head: head, tail: tail, content: content, err: err,
|
||||
}:
|
||||
case <-ctx.Done():
|
||||
return
|
||||
@@ -942,55 +1151,151 @@ func hashWorker(ctx context.Context, jobs <-chan []fileRec,
|
||||
}
|
||||
}
|
||||
|
||||
// emptyHash is the lowercase-hex SHA-256 of the empty input: the head
|
||||
// and tail hash of every zero-length file.
|
||||
// emptyHash is the lowercase-hex SHA-256 of the empty input: the head,
|
||||
// tail, and content hash of every zero-length file.
|
||||
const emptyHash = "e3b0c44298fc1c149afbf4c8996fb924" +
|
||||
"27ae41e4649b934ca495991b7852b855"
|
||||
|
||||
// hashHeadTail returns the lowercase-hex SHA-256 of the first
|
||||
// min(chunk, size) bytes and of the last min(chunk, size) bytes of the
|
||||
// file at path. The two reads overlap when size < 2*chunk. size is the
|
||||
// value recorded when the file was statted; a zero-length file's
|
||||
// hashes are constant, so it is never even opened.
|
||||
func hashHeadTail(path string, size int64) (string, string, error) {
|
||||
// hashSignature computes the hashes the hash phase records for a file
|
||||
// whose size is shared; with the file size they form its duplicate
|
||||
// signature. A file below headTailMin is hashed in full and its
|
||||
// whole-file SHA-256 is returned as head, tail, and content alike —
|
||||
// that range takes no separate end-window step. For a file at or above
|
||||
// headTailMin only the head and tail are computed, the SHA-256 of its
|
||||
// first and last headTailWindow bytes, and content is returned empty:
|
||||
// the content phase computes it with hashContentOnly once the file's
|
||||
// size, head, and tail match another file's. Two files are duplicates
|
||||
// only when all four agree; any mismatch means not a duplicate. size
|
||||
// is the value recorded when the file was statted; a zero-length file
|
||||
// has constant hashes and is never opened.
|
||||
func hashSignature(path string, size int64) (string, string, string, error) {
|
||||
if size == 0 {
|
||||
return emptyHash, emptyHash, nil
|
||||
return emptyHash, emptyHash, emptyHash, nil
|
||||
}
|
||||
|
||||
//nolint:gosec // hashing operator-supplied paths is the tool's purpose
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
return "", "", "", err
|
||||
}
|
||||
|
||||
defer func() { _ = f.Close() }()
|
||||
|
||||
n := min(int64(chunk), size)
|
||||
// Below the threshold the whole file is hashed directly, with no
|
||||
// end-window step: head and tail both carry the whole-file hash.
|
||||
if size < int64(headTailMin) {
|
||||
content, err := hashWhole(f, size)
|
||||
if err != nil {
|
||||
return "", "", "", err
|
||||
}
|
||||
|
||||
buf := make([]byte, n)
|
||||
return content, content, content, nil
|
||||
}
|
||||
|
||||
_, err = f.ReadAt(buf, 0)
|
||||
head, tail, err := hashEnds(f, size)
|
||||
if err != nil {
|
||||
return "", "", "", err
|
||||
}
|
||||
|
||||
return head, tail, "", nil
|
||||
}
|
||||
|
||||
// hashContentOnly returns the content hash of the file at path, which
|
||||
// is at least headTailMin bytes: the content phase's read. head and
|
||||
// tail are returned empty, because the content phase keeps the ones its
|
||||
// records already hold.
|
||||
func hashContentOnly(path string, size int64) (string, string, string, error) {
|
||||
//nolint:gosec // hashing operator-supplied paths is the tool's purpose
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return "", "", "", err
|
||||
}
|
||||
|
||||
defer func() { _ = f.Close() }()
|
||||
|
||||
content, err := hashContent(f, size)
|
||||
|
||||
return "", "", content, err
|
||||
}
|
||||
|
||||
// hashEnds returns the SHA-256 of the first and last headTailWindow
|
||||
// bytes of f. It is called only for files at least headTailMin, which
|
||||
// is far larger than two windows, so the windows never overlap and both
|
||||
// reads are always full.
|
||||
func hashEnds(f *os.File, size int64) (string, string, error) {
|
||||
buf := make([]byte, headTailWindow)
|
||||
|
||||
_, err := f.ReadAt(buf, 0)
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
}
|
||||
|
||||
h := sha256.Sum256(buf)
|
||||
head := hex.EncodeToString(h[:])
|
||||
|
||||
// When the whole file fits in one chunk the tail window is exactly
|
||||
// the bytes just read: reuse the head hash instead of issuing a
|
||||
// second read for every small file.
|
||||
if size <= int64(chunk) {
|
||||
hh := hex.EncodeToString(h[:])
|
||||
|
||||
return hh, hh, nil
|
||||
}
|
||||
|
||||
_, err = f.ReadAt(buf, size-n)
|
||||
_, err = f.ReadAt(buf, size-int64(headTailWindow))
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
}
|
||||
|
||||
t := sha256.Sum256(buf)
|
||||
|
||||
return hex.EncodeToString(h[:]), hex.EncodeToString(t[:]), nil
|
||||
return head, hex.EncodeToString(t[:]), nil
|
||||
}
|
||||
|
||||
// hashContent returns the content-rung hash of f: the SHA-256 of the
|
||||
// whole file when it is smaller than wholeFileMax, or of sampled
|
||||
// windows when it is that size or larger.
|
||||
func hashContent(f *os.File, size int64) (string, error) {
|
||||
if size >= int64(wholeFileMax) {
|
||||
return hashSamples(f, size)
|
||||
}
|
||||
|
||||
return hashWhole(f, size)
|
||||
}
|
||||
|
||||
// hashWhole returns the SHA-256 of the entire file. A SectionReader is
|
||||
// used so the read is independent of the offset left by any end-window
|
||||
// reads. Reading fewer than size bytes means the file shrank between
|
||||
// the stat and the hash; that is an error rather than a hash of content
|
||||
// that no longer matches the recorded size.
|
||||
func hashWhole(f *os.File, size int64) (string, error) {
|
||||
h := sha256.New()
|
||||
|
||||
n, err := io.Copy(h, io.NewSectionReader(f, 0, size))
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
if n != size {
|
||||
return "", fmt.Errorf("read %d of %d bytes: %w", n, size,
|
||||
io.ErrUnexpectedEOF)
|
||||
}
|
||||
|
||||
return hex.EncodeToString(h.Sum(nil)), nil
|
||||
}
|
||||
|
||||
// hashSamples feeds sampleWindow bytes at each gigabyte-aligned offset
|
||||
// (0, sampleStride, 2*sampleStride, ... while inside the file), in
|
||||
// order, into one hash, each window truncated at end of file. This is
|
||||
// the probabilistic large-file rung: two files of equal size agreeing
|
||||
// on every sample are reported as duplicates without every byte being
|
||||
// read. Because size is part of the signature, files of different sizes
|
||||
// never reach this comparison, so the sample boundaries always align.
|
||||
func hashSamples(f *os.File, size int64) (string, error) {
|
||||
h := sha256.New()
|
||||
buf := make([]byte, sampleWindow)
|
||||
|
||||
for off := int64(0); off < size; off += int64(sampleStride) {
|
||||
n := min(int64(sampleWindow), size-off)
|
||||
|
||||
_, err := f.ReadAt(buf[:n], off)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
h.Write(buf[:n])
|
||||
}
|
||||
|
||||
return hex.EncodeToString(h.Sum(nil)), nil
|
||||
}
|
||||
|
||||
+636
-40
@@ -53,7 +53,23 @@ func pattern(tag byte, n int) []byte {
|
||||
return data
|
||||
}
|
||||
|
||||
func TestHashHeadTail(t *testing.T) {
|
||||
// sig returns a file's full signature (head, tail, content), failing the
|
||||
// test on any error.
|
||||
func sig(t *testing.T, path string, size int64) (string, string, string) {
|
||||
t.Helper()
|
||||
|
||||
head, tail, content, err := hashSignature(path, size)
|
||||
if err != nil {
|
||||
t.Fatalf("hashSignature %s: %v", path, err)
|
||||
}
|
||||
|
||||
return head, tail, content
|
||||
}
|
||||
|
||||
// TestHashSignatureBelowThreshold verifies that a file below headTailMin
|
||||
// is hashed in full and compared directly: head, tail, and content all
|
||||
// carry the whole-file SHA-256, with no separate end-window step.
|
||||
func TestHashSignatureBelowThreshold(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
@@ -62,13 +78,10 @@ func TestHashHeadTail(t *testing.T) {
|
||||
name string
|
||||
data []byte
|
||||
}{
|
||||
{"empty", nil},
|
||||
{"one-byte", []byte("x")},
|
||||
{"under-one-chunk", pattern(1, chunk-1)},
|
||||
{"exactly-one-chunk", pattern(2, chunk)},
|
||||
{"overlapping-reads", pattern(3, chunk+chunk/2)},
|
||||
{"exactly-two-chunks", pattern(4, 2*chunk)},
|
||||
{"beyond-two-chunks", pattern(5, 3*chunk)},
|
||||
{"one-window", pattern(1, headTailWindow)},
|
||||
{"several-windows", pattern(2, 3*headTailWindow)},
|
||||
{"near-threshold", pattern(3, headTailMin-1)},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
@@ -76,49 +89,619 @@ func TestHashHeadTail(t *testing.T) {
|
||||
|
||||
p := writeFile(t, dir, c.name, c.data)
|
||||
|
||||
head, tail, err := hashHeadTail(p, int64(len(c.data)))
|
||||
if err != nil {
|
||||
t.Fatalf("hashHeadTail: %v", err)
|
||||
}
|
||||
head, tail, content := sig(t, p, int64(len(c.data)))
|
||||
|
||||
n := min(chunk, len(c.data))
|
||||
if want := hexSum(c.data[:n]); head != want {
|
||||
t.Errorf("head = %s, want %s", head, want)
|
||||
}
|
||||
|
||||
if want := hexSum(c.data[len(c.data)-n:]); tail != want {
|
||||
t.Errorf("tail = %s, want %s", tail, want)
|
||||
whole := hexSum(c.data)
|
||||
if head != whole || tail != whole || content != whole {
|
||||
t.Errorf("head=%s tail=%s content=%s, want all whole-file %s",
|
||||
head, tail, content, whole)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestHashHeadTailErrors(t *testing.T) {
|
||||
// TestHashSignatureEnds exercises the head and tail rungs, which apply
|
||||
// only to files at least headTailMin. Sparse files keep the fixtures
|
||||
// cheap: a difference in the first window changes only head, a
|
||||
// difference in the last window changes only tail, and a difference
|
||||
// between the windows changes neither end hash but does change the
|
||||
// whole-file content rung (the file is below wholeFileMax).
|
||||
// hashSignature leaves the content hash of a file this size to the
|
||||
// content phase, so that rung is checked through a scan.
|
||||
func TestHashSignatureEnds(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
|
||||
_, _, err := hashHeadTail(filepath.Join(dir, "missing"), 1)
|
||||
// Between headTailMin and wholeFileMax: the end-window gate is active
|
||||
// and the content rung is a whole-file hash.
|
||||
const size = int64(headTailMin + 2*1024*1024)
|
||||
|
||||
base := sparseFile(t, dir, "ends-base", size)
|
||||
headDiff := sparseFile(t, dir, "ends-head", size)
|
||||
tailDiff := sparseFile(t, dir, "ends-tail", size)
|
||||
midDiff := sparseFile(t, dir, "ends-mid", size)
|
||||
|
||||
pokeAt(t, headDiff, 0, []byte{1})
|
||||
pokeAt(t, tailDiff, size-1, []byte{1})
|
||||
pokeAt(t, midDiff, size/2, []byte{1})
|
||||
|
||||
bHead, bTail, bContent := sig(t, base, size)
|
||||
if bContent != "" {
|
||||
t.Errorf("content = %q, want none from the hash phase", bContent)
|
||||
}
|
||||
|
||||
h, tl, _ := sig(t, headDiff, size)
|
||||
if h == bHead {
|
||||
t.Error("a byte in the first window did not change head")
|
||||
}
|
||||
|
||||
if tl != bTail {
|
||||
t.Error("a byte in the first window changed tail")
|
||||
}
|
||||
|
||||
h, tl, _ = sig(t, tailDiff, size)
|
||||
if tl == bTail {
|
||||
t.Error("a byte in the last window did not change tail")
|
||||
}
|
||||
|
||||
if h != bHead {
|
||||
t.Error("a byte in the last window changed head")
|
||||
}
|
||||
|
||||
h, tl, _ = sig(t, midDiff, size)
|
||||
if h != bHead || tl != bTail {
|
||||
t.Error("a byte between the windows changed an end hash")
|
||||
}
|
||||
|
||||
// base and midDiff match on size, head, and tail, so the scan reads
|
||||
// both for their content hashes.
|
||||
c := scanContents(t, dir, base, midDiff)
|
||||
if c[midDiff] == c[base] {
|
||||
t.Error("whole-file content rung ignored a byte between the windows")
|
||||
}
|
||||
}
|
||||
|
||||
func TestHashSignatureErrors(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
|
||||
// A missing file: an error, and every hash left empty.
|
||||
head, tail, content, err := hashSignature(filepath.Join(dir, "missing"), 1)
|
||||
if err == nil {
|
||||
t.Error("no error for a missing file")
|
||||
}
|
||||
|
||||
if head != "" || tail != "" || content != "" {
|
||||
t.Errorf("missing file returned hashes: %q %q %q", head, tail, content)
|
||||
}
|
||||
|
||||
// A zero-length file has constant hashes and is never opened: even
|
||||
// a missing path succeeds.
|
||||
head, tail, err := hashHeadTail(filepath.Join(dir, "missing"), 0)
|
||||
if err != nil || head != emptyHash || tail != emptyHash {
|
||||
t.Errorf("empty: head=%q tail=%q err=%v, want constant hashes",
|
||||
head, tail, err)
|
||||
head, tail, content, err = hashSignature(filepath.Join(dir, "missing"), 0)
|
||||
if err != nil ||
|
||||
head != emptyHash || tail != emptyHash || content != emptyHash {
|
||||
t.Errorf("empty: head=%q tail=%q content=%q err=%v, "+
|
||||
"want constant hashes", head, tail, content, err)
|
||||
}
|
||||
|
||||
// A file that shrank between the stat and hash passes: reading at
|
||||
// the stat-reported size must fail rather than emit wrong hashes.
|
||||
p := writeFile(t, dir, "shrunk", []byte("tiny"))
|
||||
|
||||
_, _, err = hashHeadTail(p, int64(2*chunk))
|
||||
head, tail, content, err = hashSignature(p, int64(2*headTailWindow))
|
||||
if err == nil {
|
||||
t.Error("no error when the stat size exceeds the file size")
|
||||
}
|
||||
|
||||
if head != "" || tail != "" || content != "" {
|
||||
t.Errorf("shrunk file returned hashes: %q %q %q", head, tail, content)
|
||||
}
|
||||
}
|
||||
|
||||
// sparseFile creates a file that is logically size bytes long without
|
||||
// allocating blocks for the hole, so multi-gigabyte cases stay cheap.
|
||||
func sparseFile(t *testing.T, dir, name string, size int64) string {
|
||||
t.Helper()
|
||||
|
||||
p := filepath.Join(dir, name)
|
||||
|
||||
f, err := os.Create(p) //nolint:gosec // test-controlled path
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = f.Truncate(size)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = f.Close()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return p
|
||||
}
|
||||
|
||||
// pokeAt writes data into an existing file at off, leaving the rest of
|
||||
// the file (a sparse hole) untouched.
|
||||
func pokeAt(t *testing.T, path string, off int64, data []byte) {
|
||||
t.Helper()
|
||||
|
||||
f, err := os.OpenFile(path, os.O_WRONLY, 0o600) //nolint:gosec // test path
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
_, err = f.WriteAt(data, off)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = f.Close()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
// scanContents scans dir into a fresh database and returns the content
|
||||
// hash recorded for each file, by path, failing the test if one of want
|
||||
// has none. A file of headTailMin or more gets a content hash only when
|
||||
// it is scanned with a file of the same size, head, and tail.
|
||||
func scanContents(t *testing.T, dir string,
|
||||
want ...string,
|
||||
) map[string]string {
|
||||
t.Helper()
|
||||
|
||||
db := openTestDB(t)
|
||||
syncTree(t, db, dir)
|
||||
|
||||
contents := make(map[string]string)
|
||||
for _, r := range dbRecords(t, db) {
|
||||
contents[r.path] = r.content
|
||||
}
|
||||
|
||||
for _, p := range want {
|
||||
if contents[p] == "" {
|
||||
t.Fatalf("%s: no content hash", p)
|
||||
}
|
||||
}
|
||||
|
||||
return contents
|
||||
}
|
||||
|
||||
// TestContentRungBoundary checks the 50 MiB boundary between the two
|
||||
// content rungs: just below it the whole file is hashed and any byte
|
||||
// difference shows; at the boundary only the gigabyte-spaced samples are
|
||||
// hashed, so a difference outside a sample window is invisible. The
|
||||
// files of each pair match on size, head, and tail, so the scan reads
|
||||
// both for their content hashes.
|
||||
func TestContentRungBoundary(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
|
||||
// A byte that lands outside the single [0, sampleWindow) sample a
|
||||
// sub-gigabyte file has, but well inside the file.
|
||||
const off = 10 * 1024 * 1024
|
||||
|
||||
// Just under the boundary: the whole-file rung sees the poked byte.
|
||||
under := int64(wholeFileMax - 1)
|
||||
underBase := sparseFile(t, dir, "under-base", under)
|
||||
underPoked := sparseFile(t, dir, "under-poked", under)
|
||||
|
||||
pokeAt(t, underPoked, off, []byte{1})
|
||||
|
||||
// At the boundary: only [0, sampleWindow) is sampled, so the poked
|
||||
// byte at off is invisible and the two content hashes match.
|
||||
at := int64(wholeFileMax)
|
||||
atBase := sparseFile(t, dir, "at-base", at)
|
||||
atPoked := sparseFile(t, dir, "at-poked", at)
|
||||
|
||||
pokeAt(t, atPoked, off, []byte{1})
|
||||
|
||||
c := scanContents(t, dir, underBase, underPoked, atBase, atPoked)
|
||||
|
||||
if c[underBase] == c[underPoked] {
|
||||
t.Error("whole-file rung ignored a byte difference below wholeFileMax")
|
||||
}
|
||||
|
||||
if c[atBase] != c[atPoked] {
|
||||
t.Error("sampled rung saw a byte outside every sample window")
|
||||
}
|
||||
}
|
||||
|
||||
// TestContentRungMultiGigabyte exercises the sampled rung across several
|
||||
// gigabytes using sparse files: a difference inside the third sample
|
||||
// window (at offset 2*sampleStride) changes the hash, while a difference
|
||||
// in the gap after it does not. The three files match on size, head,
|
||||
// and tail, so the scan reads each for its content hash.
|
||||
func TestContentRungMultiGigabyte(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
|
||||
// Three sample windows (offsets 0, 1 GiB, 2 GiB) plus a trailing gap
|
||||
// that no sample covers.
|
||||
size := int64(2*sampleStride + 2*sampleWindow)
|
||||
thirdSample := int64(2 * sampleStride)
|
||||
gap := thirdSample + int64(sampleWindow)
|
||||
|
||||
base := sparseFile(t, dir, "g-base", size)
|
||||
inSample := sparseFile(t, dir, "g-insample", size)
|
||||
inGap := sparseFile(t, dir, "g-ingap", size)
|
||||
|
||||
pokeAt(t, inSample, thirdSample, []byte{1})
|
||||
pokeAt(t, inGap, gap, []byte{1})
|
||||
|
||||
c := scanContents(t, dir, base, inSample, inGap)
|
||||
|
||||
if c[inSample] == c[base] {
|
||||
t.Error("sample at 2 GiB was not read: difference there was invisible")
|
||||
}
|
||||
|
||||
if c[inGap] != c[base] {
|
||||
t.Error("a byte in an unsampled gap changed the content hash")
|
||||
}
|
||||
}
|
||||
|
||||
// sparseFileWithoutMatch writes name in dir as a sparse file of size
|
||||
// bytes, next to another file of that size whose first byte differs. A
|
||||
// scan then reads the file's head and tail, since its size is shared,
|
||||
// but finds no file matching them, so it gets no content hash.
|
||||
func sparseFileWithoutMatch(t *testing.T, dir, name string,
|
||||
size int64,
|
||||
) string {
|
||||
t.Helper()
|
||||
|
||||
p := sparseFile(t, dir, name, size)
|
||||
other := sparseFile(t, dir, name+"-other-head", size)
|
||||
|
||||
pokeAt(t, other, 0, []byte{1})
|
||||
|
||||
return p
|
||||
}
|
||||
|
||||
// TestScanContentGate checks that a file of headTailMin or more is read
|
||||
// for its content hash only when its size, head, and tail match another
|
||||
// file's: a same-size pair whose heads differ and one whose tails differ
|
||||
// get no content hash and are not reported, while an identical pair is
|
||||
// read and reported.
|
||||
func TestScanContentGate(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
db := openTestDB(t)
|
||||
|
||||
// Three sizes, so that no pair meets another.
|
||||
headA := sparseFile(t, dir, "head-a", headTailMin)
|
||||
headB := sparseFile(t, dir, "head-b", headTailMin)
|
||||
tailA := sparseFile(t, dir, "tail-a", headTailMin+1)
|
||||
tailB := sparseFile(t, dir, "tail-b", headTailMin+1)
|
||||
same := []string{
|
||||
sparseFile(t, dir, "same-a", headTailMin+2),
|
||||
sparseFile(t, dir, "same-b", headTailMin+2),
|
||||
}
|
||||
|
||||
pokeAt(t, headB, 0, []byte{1})
|
||||
pokeAt(t, tailB, headTailMin, []byte{1}) // its last byte
|
||||
|
||||
syncTree(t, db, dir)
|
||||
|
||||
recs := dbRecords(t, db)
|
||||
for _, p := range []string{headA, headB, tailA, tailB} {
|
||||
r := recordByPath(t, recs, p)
|
||||
if r.head == "" || r.tail == "" || r.content != "" {
|
||||
t.Errorf("%s: head = %q tail = %q content = %q, "+
|
||||
"want head and tail only", p, r.head, r.tail, r.content)
|
||||
}
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, same) {
|
||||
t.Fatalf("groups = %+v, want only the identical pair %q",
|
||||
groups, same)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentAcrossOperands checks that a stored file gets its
|
||||
// content hash when a later scan of a separate operand brings its
|
||||
// match: tree A's file has a head and tail but no content hash until
|
||||
// tree B, holding an identical file, is scanned.
|
||||
func TestScanContentAcrossOperands(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := openTestDB(t)
|
||||
a := sparseFileWithoutMatch(t, t.TempDir(), "a", headTailMin)
|
||||
|
||||
syncTree(t, db, filepath.Dir(a))
|
||||
|
||||
if r := recordByPath(t, dbRecords(t, db), a); r.head == "" || r.content != "" {
|
||||
t.Fatalf("after scanning A: %+v, want head and tail only", r)
|
||||
}
|
||||
|
||||
b := sparseFile(t, t.TempDir(), "b", headTailMin)
|
||||
|
||||
syncTree(t, db, filepath.Dir(b))
|
||||
|
||||
recs := dbRecords(t, db)
|
||||
if r := recordByPath(t, recs, a); r.content == "" {
|
||||
t.Fatalf("after scanning B: %+v, want A's file content-hashed", r)
|
||||
}
|
||||
|
||||
want := []string{a, b}
|
||||
slices.Sort(want)
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentWithinOperand checks that a rescan adding a match next
|
||||
// to an unchanged stored file gives the stored file its content hash,
|
||||
// though the hash phase leaves it alone as unchanged.
|
||||
func TestScanContentWithinOperand(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
db := openTestDB(t)
|
||||
stored := sparseFileWithoutMatch(t, dir, "d1", headTailMin)
|
||||
|
||||
syncTree(t, db, dir)
|
||||
|
||||
added := sparseFile(t, dir, "d2", headTailMin)
|
||||
|
||||
st := syncTree(t, db, dir)
|
||||
if st != (scanStats{added: 1, unchanged: 2}) {
|
||||
t.Fatalf("rescan stats = %+v, want 1 added 2 unchanged", st)
|
||||
}
|
||||
|
||||
want := []string{stored, added}
|
||||
|
||||
groups := collectDupeGroups(dbRecords(t, db))
|
||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentStalePartners checks that a stored file outside the
|
||||
// operand that has vanished, or changed, since it was recorded is not
|
||||
// read, and that its match inside the operand is not read either: the
|
||||
// match has no other partner left, so neither gets a content hash and
|
||||
// no duplicate is reported.
|
||||
func TestScanContentStalePartners(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := openTestDB(t)
|
||||
dirA := t.TempDir()
|
||||
gone := sparseFileWithoutMatch(t, dirA, "gone", headTailMin)
|
||||
changed := sparseFileWithoutMatch(t, dirA, "changed", headTailMin+1)
|
||||
|
||||
syncTree(t, db, dirA)
|
||||
|
||||
before := dbRecords(t, db)
|
||||
|
||||
err := os.Remove(gone)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
future := time.Now().Add(time.Hour)
|
||||
|
||||
err = os.Chtimes(changed, future, future)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
dirB := t.TempDir()
|
||||
sparseFile(t, dirB, "gone-copy", headTailMin)
|
||||
sparseFile(t, dirB, "changed-copy", headTailMin+1)
|
||||
|
||||
st := syncTree(t, db, dirB)
|
||||
if st != (scanStats{added: 2}) {
|
||||
t.Errorf("stats = %+v, want 2 added and nothing skipped", st)
|
||||
}
|
||||
|
||||
recs := dbRecords(t, db)
|
||||
for _, r := range recs {
|
||||
if r.content != "" {
|
||||
t.Errorf("%s: content = %q, want none: its only match is stale",
|
||||
r.path, r.content)
|
||||
}
|
||||
}
|
||||
|
||||
for _, old := range before {
|
||||
if r := recordByPath(t, recs, old.path); r != old {
|
||||
t.Errorf("record = %+v, want it left as %+v", r, old)
|
||||
}
|
||||
}
|
||||
|
||||
if groups := collectDupeGroups(recs); len(groups) != 0 {
|
||||
t.Errorf("groups = %+v, want none", groups)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentHashedStalePartners checks that stored matches outside
|
||||
// the operand that already have a content hash are checked like any
|
||||
// other: once one has vanished and the other has changed, a copy of
|
||||
// them scanned in another tree has no match left, so it is not read and
|
||||
// is not reported as their duplicate.
|
||||
func TestScanContentHashedStalePartners(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := openTestDB(t)
|
||||
dirA := t.TempDir()
|
||||
stored := []string{
|
||||
sparseFile(t, dirA, "changed", headTailMin),
|
||||
sparseFile(t, dirA, "gone", headTailMin),
|
||||
}
|
||||
|
||||
// The two stored files match, so this scan gives both a content
|
||||
// hash.
|
||||
syncTree(t, db, dirA)
|
||||
|
||||
err := os.Remove(stored[1])
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
future := time.Now().Add(time.Hour)
|
||||
|
||||
err = os.Chtimes(stored[0], future, future)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
b := sparseFile(t, t.TempDir(), "copy", headTailMin)
|
||||
|
||||
st := syncTree(t, db, filepath.Dir(b))
|
||||
if st != (scanStats{added: 1}) {
|
||||
t.Errorf("stats = %+v, want 1 added and nothing skipped", st)
|
||||
}
|
||||
|
||||
recs := dbRecords(t, db)
|
||||
if r := recordByPath(t, recs, b); r.content != "" {
|
||||
t.Errorf("copy: content = %q, want none: its only matches are stale",
|
||||
r.content)
|
||||
}
|
||||
|
||||
// The stored records lie outside the operand and are left as they
|
||||
// are, so they still group with each other, but not with the copy.
|
||||
groups := collectDupeGroups(recs)
|
||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) {
|
||||
t.Errorf("groups = %+v, want only the stored pair %q", groups, stored)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentReadFailure checks that a failed content read is
|
||||
// counted as skipped and leaves the record without a content hash, and
|
||||
// that a later scan tries the read again.
|
||||
func TestScanContentReadFailure(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := openTestDB(t)
|
||||
a := sparseFileWithoutMatch(t, t.TempDir(), "a", headTailMin)
|
||||
|
||||
syncTree(t, db, filepath.Dir(a))
|
||||
|
||||
// lstat still works on the unreadable file, so it passes the check
|
||||
// and fails only when it is read.
|
||||
err := os.Chmod(a, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
dirB := t.TempDir()
|
||||
b := sparseFile(t, dirB, "b", headTailMin)
|
||||
|
||||
st := syncTree(t, db, dirB)
|
||||
if st != (scanStats{added: 1, skipped: 1}) {
|
||||
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
||||
}
|
||||
|
||||
if r := recordByPath(t, dbRecords(t, db), a); r.content != "" {
|
||||
t.Fatalf("unreadable file: %+v, want no content hash", r)
|
||||
}
|
||||
|
||||
err = os.Chmod(a, 0o600)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
st = syncTree(t, db, dirB)
|
||||
if st != (scanStats{unchanged: 1}) {
|
||||
t.Fatalf("rescan stats = %+v, want 1 unchanged", st)
|
||||
}
|
||||
|
||||
want := []string{a, b}
|
||||
slices.Sort(want)
|
||||
|
||||
groups := collectDupeGroups(dbRecords(t, db))
|
||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||
t.Fatalf("groups = %+v, want the pair %q after the retry",
|
||||
groups, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentCheckError checks that a stored file the content phase
|
||||
// cannot lstat, for a reason other than its being gone, is counted as
|
||||
// skipped and does not count as a match.
|
||||
func TestScanContentCheckError(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := openTestDB(t)
|
||||
sub := filepath.Join(t.TempDir(), "sub")
|
||||
|
||||
err := os.Mkdir(sub, 0o700)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
sparseFileWithoutMatch(t, sub, "a", headTailMin)
|
||||
syncTree(t, db, sub)
|
||||
|
||||
// Without search permission on its directory, the stored file's
|
||||
// lstat fails with permission denied.
|
||||
err = os.Chmod(sub, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
t.Cleanup(func() {
|
||||
//nolint:gosec // removing the directory needs its search bit back
|
||||
_ = os.Chmod(sub, 0o700)
|
||||
})
|
||||
|
||||
b := sparseFile(t, t.TempDir(), "b", headTailMin)
|
||||
|
||||
st := syncTree(t, db, filepath.Dir(b))
|
||||
if st != (scanStats{added: 1, skipped: 1}) {
|
||||
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
||||
}
|
||||
|
||||
if r := recordByPath(t, dbRecords(t, db), b); r.content != "" {
|
||||
t.Errorf("b: content = %q, want none: its only match could not be "+
|
||||
"checked", r.content)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentHardlinks checks that the content phase stores the
|
||||
// content hash of a hard-linked file on every one of its links.
|
||||
func TestScanContentHardlinks(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
db := openTestDB(t)
|
||||
a := sparseFile(t, dir, "a", headTailMin)
|
||||
b := filepath.Join(dir, "b")
|
||||
|
||||
err := os.Link(a, b)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
c := sparseFile(t, dir, "copy", headTailMin)
|
||||
|
||||
st := syncTree(t, db, dir)
|
||||
if st != (scanStats{added: 3}) {
|
||||
t.Fatalf("stats = %+v, want 3 added", st)
|
||||
}
|
||||
|
||||
recs := dbRecords(t, db)
|
||||
|
||||
want := recordByPath(t, recs, c).content
|
||||
if want == "" {
|
||||
t.Fatal("the copy has no content hash")
|
||||
}
|
||||
|
||||
for _, p := range []string{a, b} {
|
||||
if got := recordByPath(t, recs, p).content; got != want {
|
||||
t.Errorf("%s: content = %q, want %q", p, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// collectWalk runs a walk over roots and returns the emitted records
|
||||
@@ -706,9 +1289,9 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
||||
|
||||
recs := dbRecords(t, db)
|
||||
for _, r := range recs {
|
||||
if r.head != "" || r.tail != "" {
|
||||
t.Errorf("%s: head = %q tail = %q, want unhashed",
|
||||
r.path, r.head, r.tail)
|
||||
if r.head != "" || r.tail != "" || r.content != "" {
|
||||
t.Errorf("%s: head = %q tail = %q content = %q, want unhashed",
|
||||
r.path, r.head, r.tail, r.content)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -740,23 +1323,36 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
||||
func TestTreesUnhashedNeverEqual(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// Two trees identical except for unhashed same-name, same-size
|
||||
// files (possible when the trees were scanned separately) must not
|
||||
// compare equal: unhashed content is unknown.
|
||||
shared := pattern(1, 100)
|
||||
recs := []scanRec{
|
||||
{path: "/x/t1/f1", size: 100, head: hexSum(shared), tail: hexSum(shared)},
|
||||
{path: "/x/t2/f1", size: 100, head: hexSum(shared), tail: hexSum(shared)},
|
||||
{path: "/x/t1/u", size: 50},
|
||||
{path: "/x/t2/u", size: 50},
|
||||
// Two trees identical except for same-name, same-size files without
|
||||
// a content hash must not compare equal: their content is unknown.
|
||||
// That holds for unhashed files (possible when the trees were
|
||||
// scanned separately) and for files of headTailMin or more that
|
||||
// have only a head and tail.
|
||||
sum := hexSum(pattern(1, 100))
|
||||
shared := []scanRec{
|
||||
{path: "/x/t1/f1", size: 100, head: sum, tail: sum, content: sum},
|
||||
{path: "/x/t2/f1", size: 100, head: sum, tail: sum, content: sum},
|
||||
}
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
cases := map[string][]scanRec{
|
||||
"unhashed": {
|
||||
{path: "/x/t1/u", size: 50},
|
||||
{path: "/x/t2/u", size: 50},
|
||||
},
|
||||
"head and tail only": {
|
||||
{path: "/x/t1/u", size: headTailMin, head: "h", tail: "t"},
|
||||
{path: "/x/t2/u", size: headTailMin, head: "h", tail: "t"},
|
||||
},
|
||||
}
|
||||
|
||||
for name, unknown := range cases {
|
||||
super, dirs := buildHierarchy(append(slices.Clone(shared), unknown...))
|
||||
super.compute()
|
||||
|
||||
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
|
||||
t.Fatalf("tree groups = %d, want 0 (unhashed files differ)",
|
||||
len(tg))
|
||||
t.Errorf("%s: tree groups = %d, want 0 (the files may differ)",
|
||||
name, len(tg))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -11,12 +11,6 @@ set -eu
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
|
||||
# yarn provides prettier, which formats Markdown. yarn is a tool, like
|
||||
# node/git/make/go below; the reference that governs formatting output is
|
||||
# prettier, pinned by yarn.lock's integrity hash and installed by
|
||||
# `yarn install --frozen-lockfile`.
|
||||
YARN_VERSION="1.22.22"
|
||||
|
||||
PKGMGR=""
|
||||
SUDO=""
|
||||
APT_UPDATED=""
|
||||
@@ -64,21 +58,6 @@ missing() {
|
||||
! command -v "$1" >/dev/null 2>&1
|
||||
}
|
||||
|
||||
ensure_node() {
|
||||
if ! missing node; then return 0; fi
|
||||
pkg_install nodejs nodejs node nodejs
|
||||
}
|
||||
|
||||
ensure_yarn() {
|
||||
if ! missing yarn; then return 0; fi
|
||||
if ! missing corepack; then
|
||||
corepack enable >/dev/null 2>&1 || true
|
||||
corepack prepare "yarn@$YARN_VERSION" --activate
|
||||
else
|
||||
pkg_install yarn yarn yarn yarn
|
||||
fi
|
||||
}
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
|
||||
@@ -92,16 +71,6 @@ main() {
|
||||
if missing make; then pkg_install gnumake make make make; fi
|
||||
if missing go; then pkg_install go golang go go; fi
|
||||
|
||||
# node runs prettier and is an unpinned host tool for the same reason
|
||||
# git/make/go are: it comes from the host package manager, whatever
|
||||
# version it ships. It is not installed via nvm the way the canonical
|
||||
# template does, because nvm's prebuilt node is glibc-linked and does
|
||||
# not run on this repo's musl/Alpine build image. prettier — the tool
|
||||
# whose version affects formatting output — is pinned by yarn.lock.
|
||||
ensure_node
|
||||
ensure_yarn
|
||||
yarn install --frozen-lockfile
|
||||
|
||||
# Linting runs via docker only (script/lint), so docker is a lint
|
||||
# prerequisite rather than something bootstrap installs. Warn, do
|
||||
# not fail: everything except `make lint` — and, through it,
|
||||
|
||||
+4
-4
@@ -16,10 +16,10 @@
|
||||
# implies the repo is green.
|
||||
#
|
||||
# That implication holds only because of CHECK_EPOCH. A COPY layer is
|
||||
# invalidated by changed content, and a merge commit's tree is
|
||||
# byte-identical to the branch head it merges, so without a fresh value
|
||||
# here Docker serves the gate layers from cache and the build reports a
|
||||
# green it never earned. Passing the current epoch invalidates the gate
|
||||
# invalidated only by changed content, and a rebuild of an unchanged
|
||||
# checkout sends the same content, so without a fresh value here Docker
|
||||
# serves the gate layers from cache and the build reports a green it
|
||||
# never earned. Passing the current epoch invalidates the gate
|
||||
# layers on every run while leaving the pinned base images and
|
||||
# go mod download cached; see the Dockerfile for the placement.
|
||||
set -eu
|
||||
|
||||
+1
-12
@@ -1,23 +1,12 @@
|
||||
#!/bin/sh
|
||||
# script/fmt: format all files (writes). gofmt for Go, prettier for
|
||||
# Markdown. prettier is the pinned devDependency in package.json/
|
||||
# yarn.lock; script/bootstrap installs it (see run_prettier).
|
||||
# script/fmt: format all files (writes).
|
||||
set -eu
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
|
||||
run_prettier() {
|
||||
if ! command -v yarn >/dev/null 2>&1; then
|
||||
echo "fmt: yarn not found; run script/bootstrap first" >&2
|
||||
exit 1
|
||||
fi
|
||||
yarn run prettier "$@"
|
||||
}
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
gofmt -s -w .
|
||||
run_prettier --write '**/*.md' --tab-width 4 --prose-wrap always
|
||||
}
|
||||
|
||||
main "$@"
|
||||
|
||||
+2
-21
@@ -1,37 +1,18 @@
|
||||
#!/bin/sh
|
||||
# script/fmt-check: check formatting (read-only). Same scope as
|
||||
# script/fmt: gofmt for Go, prettier for Markdown. Both run every time
|
||||
# and each reports independently, so a failure names which formatter is
|
||||
# unhappy; the script exits non-zero if either found unformatted files.
|
||||
# script/fmt, but fails instead of writing.
|
||||
set -eu
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
|
||||
run_prettier() {
|
||||
if ! command -v yarn >/dev/null 2>&1; then
|
||||
echo "fmt-check: yarn not found; run script/bootstrap first" >&2
|
||||
exit 1
|
||||
fi
|
||||
yarn run prettier "$@"
|
||||
}
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
rc=0
|
||||
|
||||
files="$(gofmt -s -l .)"
|
||||
if [ -n "$files" ]; then
|
||||
echo "gofmt: files not formatted:" >&2
|
||||
echo "$files" >&2
|
||||
rc=1
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ! run_prettier --check '**/*.md' --tab-width 4 --prose-wrap always; then
|
||||
echo "prettier: Markdown not formatted; run make fmt" >&2
|
||||
rc=1
|
||||
fi
|
||||
|
||||
exit "$rc"
|
||||
}
|
||||
|
||||
main "$@"
|
||||
|
||||
@@ -16,6 +16,7 @@ type fileSig struct {
|
||||
size int64
|
||||
head string
|
||||
tail string
|
||||
content string
|
||||
}
|
||||
|
||||
// treeNode is one directory reconstructed from the scan stream.
|
||||
@@ -122,14 +123,16 @@ func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
||||
node.files = make(map[string]fileSig)
|
||||
}
|
||||
|
||||
sig := fileSig{size: r.size, head: r.head, tail: r.tail}
|
||||
sig := fileSig{
|
||||
size: r.size, head: r.head, tail: r.tail, content: r.content,
|
||||
}
|
||||
|
||||
// An unhashed record (its size was unique when last scanned)
|
||||
// has unknown content: give it a signature no other file can
|
||||
// share, so trees containing it never compare equal. Real
|
||||
// heads are hex, so the NUL-prefixed form cannot collide.
|
||||
if sig.head == "" {
|
||||
sig.head = "unhashed\x00" + r.path
|
||||
// A record without a content hash has unknown content (README
|
||||
// "Database"): give it a signature no other file can share, so
|
||||
// trees containing it never compare equal. Real hashes are
|
||||
// hex, so the NUL-prefixed form cannot collide.
|
||||
if sig.content == "" {
|
||||
sig.content = "unhashed\x00" + r.path
|
||||
}
|
||||
|
||||
node.files[comps[len(comps)-1]] = sig
|
||||
@@ -188,7 +191,7 @@ func (n *treeNode) compute() {
|
||||
for name, sig := range n.files {
|
||||
entries = append(entries,
|
||||
"f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+
|
||||
"\x00"+sig.head+"\x00"+sig.tail)
|
||||
"\x00"+sig.head+"\x00"+sig.tail+"\x00"+sig.content)
|
||||
n.fileCount++
|
||||
n.totalSize += sig.size
|
||||
}
|
||||
|
||||
+16
-13
@@ -9,20 +9,23 @@ import (
|
||||
const (
|
||||
f1Head = "f1h"
|
||||
f1Tail = "f1t"
|
||||
f1Content = "f1c"
|
||||
f2Head = "f2h"
|
||||
f2Tail = "f2t"
|
||||
f2Content = "f2c"
|
||||
)
|
||||
|
||||
// smokeTreeRecs mirrors the README smoke-test tree layout: /d/t1 and
|
||||
// /d/t2 are identical, /d/t3 differs from them only by one filename.
|
||||
func smokeTreeRecs() []scanRec {
|
||||
return []scanRec{
|
||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t1/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t1/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t2/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t2/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t3/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t3/sub/f2renamed"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t1/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t1/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t2/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t2/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t3/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, content: f2Content,
|
||||
path: "/d/t3/sub/f2renamed"},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -114,8 +117,8 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
|
||||
const sharedTail = "same"
|
||||
|
||||
recs := []scanRec{
|
||||
{size: 10, head: sharedTail, tail: sharedTail, path: "/r/a/f"},
|
||||
{size: 10, head: "DIFF", tail: sharedTail, path: "/r/b/f"},
|
||||
{size: 10, head: sharedTail, tail: sharedTail, content: "c", path: "/r/a/f"},
|
||||
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
|
||||
}
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
@@ -181,8 +184,8 @@ func TestCollectTreeGroupsSiblings(t *testing.T) {
|
||||
// Identical sibling dirs share a parent, so their group cannot be
|
||||
// implied by a parent group and must be reported.
|
||||
recs := []scanRec{
|
||||
{size: 10, head: "h", tail: "t", path: "/p/x1/f"},
|
||||
{size: 10, head: "h", tail: "t", path: "/p/x2/f"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x1/f"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
|
||||
}
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
@@ -203,9 +206,9 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) {
|
||||
// extra file, so the parents' digests differ and the x group must
|
||||
// be reported.
|
||||
recs := []scanRec{
|
||||
{size: 10, head: "h", tail: "t", path: "/p/a/x/f"},
|
||||
{size: 99, head: "e", tail: "e", path: "/p/a/extra"},
|
||||
{size: 10, head: "h", tail: "t", path: "/q/b/x/f"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/a/x/f"},
|
||||
{size: 99, head: "e", tail: "e", content: "e", path: "/p/a/extra"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
|
||||
}
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
|
||||
@@ -1,8 +0,0 @@
|
||||
# THIS IS AN AUTOGENERATED FILE. DO NOT EDIT THIS FILE DIRECTLY.
|
||||
# yarn lockfile v1
|
||||
|
||||
|
||||
prettier@3.8.1:
|
||||
version "3.8.1"
|
||||
resolved "https://registry.yarnpkg.com/prettier/-/prettier-3.8.1.tgz#edf48977cf991558f4fcbd8a3ba6015ba2a3a173"
|
||||
integrity sha512-UOnG6LftzbdaHZcKoPFtOcCKztrQ57WkHDeRD9t/PTQtmT0NHSeWWepj6pS0z/N7+08BHFDQVUrfmfMRcZwbMg==
|
||||
Reference in New Issue
Block a user