Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1f64f4d6de | ||
|
|
c737490a53 | ||
|
|
09a39ddf37 | ||
|
|
29a65016d0 | ||
|
|
7ac4f6b723 |
+6
-3
@@ -1,9 +1,12 @@
|
|||||||
.git
|
# .git is sent without its config. Without a VERSION build argument the
|
||||||
|
# stage that compiles runs `git describe --tags --always` on .git, which
|
||||||
|
# does not need .git/config; that file can hold a credential, such as a
|
||||||
|
# password in a remote URL or the token the CI checkout step stores there.
|
||||||
|
.git/config
|
||||||
|
|
||||||
.claude
|
.claude
|
||||||
.DS_Store
|
.DS_Store
|
||||||
sfdupes
|
sfdupes
|
||||||
files.dat
|
|
||||||
node_modules
|
|
||||||
*.log
|
*.log
|
||||||
*.out
|
*.out
|
||||||
*.test
|
*.test
|
||||||
|
|||||||
@@ -6,4 +6,6 @@ jobs:
|
|||||||
steps:
|
steps:
|
||||||
# actions/checkout v4.2.2, 2026-02-22
|
# actions/checkout v4.2.2, 2026-02-22
|
||||||
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
|
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
- run: script/cibuild
|
- run: script/cibuild
|
||||||
|
|||||||
@@ -27,7 +27,6 @@ node_modules/
|
|||||||
*.log
|
*.log
|
||||||
|
|
||||||
# Local scan data
|
# Local scan data
|
||||||
files.dat
|
|
||||||
*.sqlite
|
*.sqlite
|
||||||
*.sqlite-shm
|
*.sqlite-shm
|
||||||
*.sqlite-wal
|
*.sqlite-wal
|
||||||
|
|||||||
@@ -1,2 +0,0 @@
|
|||||||
node_modules/
|
|
||||||
yarn.lock
|
|
||||||
@@ -1,4 +0,0 @@
|
|||||||
{
|
|
||||||
"tabWidth": 4,
|
|
||||||
"proseWrap": "always"
|
|
||||||
}
|
|
||||||
+19
-12
@@ -27,12 +27,9 @@ ARG CHECK_EPOCH
|
|||||||
# target now runs `docker build -f Dockerfile.lint`, and a docker build
|
# target now runs `docker build -f Dockerfile.lint`, and a docker build
|
||||||
# cannot run a docker build: routing the gate through make would mean
|
# cannot run a docker build: routing the gate through make would mean
|
||||||
# nesting docker inside this image. Same reason `make check` is gone
|
# nesting docker inside this image. Same reason `make check` is gone
|
||||||
# from the build stage below.
|
# from the build stage below. `make fmt-check` stays as it is — it is a
|
||||||
#
|
# gate, not the aggregate, and it shells out to nothing.
|
||||||
# `make fmt-check` is not run in this stage: it now also runs prettier
|
RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
|
||||||
# over Markdown, and this golangci-lint image has no node. The gate runs
|
|
||||||
# in the build stage below, where script/bootstrap installs node and
|
|
||||||
# prettier.
|
|
||||||
|
|
||||||
# The FROM above and the one in Dockerfile.lint pin the same linter
|
# The FROM above and the one in Dockerfile.lint pin the same linter
|
||||||
# twice, and nothing else keeps them in sync; this fails the build when
|
# twice, and nothing else keeps them in sync; this fails the build when
|
||||||
@@ -80,12 +77,10 @@ COPY --from=lint /src/go.sum /dev/null
|
|||||||
# rather than duplicating the installs inline. Only script/ and the
|
# rather than duplicating the installs inline. Only script/ and the
|
||||||
# dependency manifests are copied first, nothing else, so this layer
|
# dependency manifests are copied first, nothing else, so this layer
|
||||||
# stays cached until the scripts or the dependencies change — bootstrap
|
# stays cached until the scripts or the dependencies change — bootstrap
|
||||||
# runs `go mod download` and `yarn install`, which is why there is no
|
# ends in `go mod download`, which is why there is no separate
|
||||||
# separate invocation of either here. The JS manifests (package.json,
|
# invocation of it here.
|
||||||
# yarn.lock) are copied too so the yarn install layer caches alongside
|
|
||||||
# the Go one.
|
|
||||||
COPY script/ script/
|
COPY script/ script/
|
||||||
COPY go.mod go.sum package.json yarn.lock ./
|
COPY go.mod go.sum ./
|
||||||
RUN script/bootstrap
|
RUN script/bootstrap
|
||||||
|
|
||||||
COPY . .
|
COPY . .
|
||||||
@@ -113,7 +108,19 @@ ARG CHECK_EPOCH
|
|||||||
RUN echo "gate test, epoch ${CHECK_EPOCH}" && make test
|
RUN echo "gate test, epoch ${CHECK_EPOCH}" && make test
|
||||||
RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
|
RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
|
||||||
|
|
||||||
RUN make build
|
# The version stamped into the binary: the VERSION build argument when
|
||||||
|
# one is given, otherwise `git describe --tags --always` of the .git in
|
||||||
|
# the build context (git is installed by script/bootstrap above). A
|
||||||
|
# context that carries .git and still yields no version fails the build;
|
||||||
|
# with neither, as from a source tarball, it is "dev".
|
||||||
|
ARG VERSION
|
||||||
|
RUN version="${VERSION:-$(git describe --tags --always || echo dev)}"; \
|
||||||
|
if [ -e .git ] && { [ -z "$version" ] || [ "$version" = dev ] || \
|
||||||
|
[ "$version" = unknown ]; }; then \
|
||||||
|
echo "no version could be derived although the build context carries .git" >&2; \
|
||||||
|
exit 1; \
|
||||||
|
fi; \
|
||||||
|
make build VERSION="$version"
|
||||||
|
|
||||||
# Runtime stage
|
# Runtime stage
|
||||||
# alpine:3.22, 2026-07-23
|
# alpine:3.22, 2026-07-23
|
||||||
|
|||||||
@@ -46,4 +46,4 @@ hooks:
|
|||||||
@script/install-precommit
|
@script/install-precommit
|
||||||
|
|
||||||
clean:
|
clean:
|
||||||
rm -f $(BINARY) files.dat
|
rm -f $(BINARY)
|
||||||
|
|||||||
@@ -1,386 +1,464 @@
|
|||||||
# Workflow
|
# Workflow
|
||||||
|
|
||||||
- take an issue from the `1.0.0` milestone on the tracker; work not yet on the
|
- take an issue from the `1.0.0` milestone on the tracker; work not
|
||||||
tracker gets filed as an issue first
|
yet on the tracker gets filed as an issue first
|
||||||
- branch (from `main`)
|
- branch (from `main`)
|
||||||
- do the work, with tests, in small focused commits
|
- do the work, with tests, in small focused commits
|
||||||
- record it at the top of Completed Steps (`TODO.md` changes in the same commit
|
- record it at the top of Completed Steps (`TODO.md` changes in the
|
||||||
as the work)
|
same commit as the work)
|
||||||
- push the branch and open a PR whose title ends with ` (closes #N)`
|
- push the branch and open a PR whose title ends with
|
||||||
- an independent review gates the merge; every finding is addressed or
|
` (closes #N)`
|
||||||
explicitly rebutted on the PR
|
- an independent review gates the merge; every finding is addressed
|
||||||
|
or explicitly rebutted on the PR
|
||||||
- merge to `main` once the review passes
|
- merge to `main` once the review passes
|
||||||
|
|
||||||
# Status
|
# Status
|
||||||
|
|
||||||
- pre-1.0
|
- pre-1.0
|
||||||
- the Gitea tracker is authoritative for the pre-1.0 backlog: the open issues
|
- the Gitea tracker is authoritative for the pre-1.0 backlog: the
|
||||||
under the `1.0.0` milestone are what remains before the tag, and this file
|
open issues under the `1.0.0` milestone are what remains before
|
||||||
records history and process, not the queue
|
the tag, and this file records history and process, not the queue
|
||||||
|
|
||||||
# Next Step
|
# Next Step
|
||||||
|
|
||||||
- take the next issue from the `1.0.0` milestone on the tracker:
|
- take the next issue from the `1.0.0` milestone on the tracker:
|
||||||
https://git.eeqj.de/sneak/sfdupes/milestone/17 — the milestone is the source
|
https://git.eeqj.de/sneak/sfdupes/milestone/17 — the milestone is
|
||||||
of truth for what is left before 1.0.0. Individual issues are deliberately not
|
the source of truth for what is left before 1.0.0. Individual
|
||||||
restated here; a copy in this file drifts out of date the moment the tracker
|
issues are deliberately not restated here; a copy in this file
|
||||||
moves
|
drifts out of date the moment the tracker moves
|
||||||
|
|
||||||
# Completed Steps
|
# Completed Steps
|
||||||
|
|
||||||
- restore Markdown formatting in `script/fmt`/`fmt-check` and reformat all
|
- stamp the git tag or short commit in a plain `docker build .`
|
||||||
Markdown to the house prettier settings (2026-09-21, closes
|
instead of `dev` (2026-10-02, branch `next`, closes
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/19)
|
https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now
|
||||||
|
sends `.git`, without `.git/config`, and the `Dockerfile` build
|
||||||
|
stage takes the `VERSION` build argument when one is given,
|
||||||
|
otherwise `git describe --tags --always` of that `.git`. The build
|
||||||
|
fails if the context carries `.git` and the version still comes out
|
||||||
|
empty, `dev` or `unknown`. The CI checkout step fetches the full
|
||||||
|
history (`fetch-depth: 0`) so CI sees the tag and stamps the same
|
||||||
|
value as `make build`.
|
||||||
|
|
||||||
|
- replace the 1 KiB end-window sampling with the head/tail plus
|
||||||
|
content-hash ladder (2026-09-22, branch `next`, closes
|
||||||
|
https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is
|
||||||
|
hashed in full and compared directly, with no end-window step — its
|
||||||
|
`head`, `tail`, and `content` all hold the whole-file hash. A file at
|
||||||
|
10 MiB or above gets only the 64 KiB `head` and `tail` in the hash
|
||||||
|
phase; a new content phase, after the update phase, reads it for its
|
||||||
|
`content` hash — the whole file below 50 MiB, gigabyte-spaced 1 MiB
|
||||||
|
samples at or above — only when its size, `head`, and `tail` match
|
||||||
|
another record's, from the same scan or stored by an earlier one, so
|
||||||
|
a stored file gains its content hash when it gains a match. A file
|
||||||
|
that is gone or has changed since its record was written is not
|
||||||
|
read. The `content` column is part of the version 1 schema. `report`
|
||||||
|
and `trees` group by the extended signature and leave out any record
|
||||||
|
without a `content` hash, so the ladder is applied across the whole
|
||||||
|
database. README "Duplicate detection" documents every rung including
|
||||||
|
the probabilistic large-file path.
|
||||||
|
|
||||||
|
- remove the dead `files.dat` references from `Makefile`, `.gitignore`
|
||||||
|
and `.dockerignore` (2026-09-21, branch `next`, closes
|
||||||
|
https://git.eeqj.de/sneak/sfdupes/issues/22)
|
||||||
|
|
||||||
- fix the lint-image pin comments and `FROM` form in `Dockerfile` and
|
- fix the lint-image pin comments and `FROM` form in `Dockerfile` and
|
||||||
`Dockerfile.lint` (2026-08-10, branch `next`, closes
|
`Dockerfile.lint` (2026-08-10, branch `next`, closes
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/25): dropped the false
|
https://git.eeqj.de/sneak/sfdupes/issues/25): dropped the false
|
||||||
`(Debian-based)` parenthetical (v2.12.1 was Debian too) and the redundant tag,
|
`(Debian-based)` parenthetical (v2.12.1 was Debian too) and the
|
||||||
so both pins are the policy `# image:vX.Y.Z, YYYY-MM-DD` comment over a bare
|
redundant tag, so both pins are the policy `# image:vX.Y.Z,
|
||||||
`FROM image@sha256:...`. Digest unchanged. `script/verify-lint-image-pin`
|
YYYY-MM-DD` comment over a bare `FROM image@sha256:...`. Digest
|
||||||
parses those `FROM` lines and still matches the tagless form; its advice line
|
unchanged. `script/verify-lint-image-pin` parses those `FROM` lines
|
||||||
lost the now meaningless "tag and digest". With no tag in either reference, a
|
and still matches the tagless form; its advice line lost the now
|
||||||
tag-only disagreement no longer exists — a one-sided tag is caught as a plain
|
meaningless "tag and digest". With no tag in either reference, a
|
||||||
mismatch.
|
tag-only disagreement no longer exists — a one-sided tag is caught as
|
||||||
|
a plain mismatch.
|
||||||
|
|
||||||
- run all linting in Docker via `Dockerfile.lint` and `script/lint` (2026-08-10,
|
- run all linting in Docker via `Dockerfile.lint` and `script/lint`
|
||||||
branch `next`, closes https://git.eeqj.de/sneak/sfdupes/issues/46): per the
|
(2026-08-10, branch `next`, closes
|
||||||
owner ruling, the linter runs inside a container invoked through the `script/`
|
https://git.eeqj.de/sneak/sfdupes/issues/46): per the owner ruling, the
|
||||||
entrypoint and is never installed on a host. New root `Dockerfile.lint` COPYs
|
linter runs inside a container invoked through the `script/`
|
||||||
the repo into the digest-pinned `golangci/golangci-lint:v2.12.2` image and
|
entrypoint and is never installed on a host. New root
|
||||||
runs `golangci-lint config verify` and `golangci-lint run` as build steps, so
|
`Dockerfile.lint` COPYs the repo into the digest-pinned
|
||||||
a successful build IS a clean lint; `script/lint` is reduced to building it.
|
`golangci/golangci-lint:v2.12.2` image and runs
|
||||||
`script/bootstrap` loses the `go install`, the pin constants, the version
|
`golangci-lint config verify` and `golangci-lint run` as build
|
||||||
parser and `verify_golangci_lint` outright rather than hardening them — with
|
steps, so a successful build IS a clean lint; `script/lint` is
|
||||||
nothing linting on the host, the `$GOPATH/bin` versus `PATH` problem that
|
reduced to building it. `script/bootstrap` loses the `go install`,
|
||||||
motivated them has no subject — and now warns rather than fails when `docker`
|
the pin constants, the version parser and `verify_golangci_lint`
|
||||||
is absent. Two traps handled. A lint build on an unchanged tree returns
|
outright rather than hardening them — with nothing linting on the
|
||||||
success in well under a second having run no linter, which is
|
host, the `$GOPATH/bin` versus `PATH` problem that motivated them has
|
||||||
|
no subject — and now warns rather than fails when `docker` is absent.
|
||||||
|
Two traps handled. A lint build on an unchanged tree returns success
|
||||||
|
in well under a second having run no linter, which is
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/32 and
|
https://git.eeqj.de/sneak/sfdupes/issues/32 and
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/39 again, so `Dockerfile.lint`
|
https://git.eeqj.de/sneak/sfdupes/issues/39 again, so
|
||||||
carries `ARG CHECK_EPOCH` referenced inside every gate `RUN` (BuildKit hashes
|
`Dockerfile.lint` carries `ARG CHECK_EPOCH` referenced
|
||||||
the expanded command, not the declaration) and `script/lint` passes
|
inside every gate `RUN` (BuildKit hashes the expanded command, not
|
||||||
`"$(date +%s)-$$"` — the PID matters because two lint runs land inside the
|
the declaration) and `script/lint` passes `"$(date +%s)-$$"` — the
|
||||||
same second easily. And nothing inside an image build may shell out to docker,
|
PID matters because two lint runs land inside the same second easily.
|
||||||
so the main `Dockerfile`'s lint stage now invokes `golangci-lint` directly
|
And nothing inside an image build may shell out to docker, so the
|
||||||
|
main `Dockerfile`'s lint stage now invokes `golangci-lint` directly
|
||||||
instead of `make lint`, and its build stage runs `make test` and
|
instead of `make lint`, and its build stage runs `make test` and
|
||||||
`make fmt-check` instead of the `make check` aggregate (`make`, not the
|
`make fmt-check` instead of the `make check` aggregate (`make`, not
|
||||||
scripts bare, because the Makefile's `export CGO_ENABLED = 0` only reaches
|
the scripts bare, because the Makefile's `export CGO_ENABLED = 0`
|
||||||
what it invokes). `COPY --from=lint` `/usr/bin/golangci-lint` is replaced by
|
only reaches what it invokes). `COPY --from=lint`
|
||||||
`COPY --from=lint /src/go.sum /dev/null`: the copied binary was the only edge
|
`/usr/bin/golangci-lint` is replaced by
|
||||||
forcing BuildKit to finish linting before the build stage starts, and dropping
|
`COPY --from=lint /src/go.sum /dev/null`: the copied binary was the
|
||||||
it without replacing the edge would have ended fail-fast linting silently
|
only edge forcing BuildKit to finish linting before the build stage
|
||||||
under a still-green build. That is canonical `REPO_POLICIES.md:107`'s ordering
|
starts, and dropping it without replacing the edge would have ended
|
||||||
edge, restored. `ENV PATH=/home/builder/go/bin:$PATH` is gone with the
|
fail-fast linting silently under a still-green build. That is
|
||||||
`go install` that justified it. `script/verify-linter-pin` is retired, deleted
|
canonical `REPO_POLICIES.md:107`'s ordering edge, restored.
|
||||||
along with its README entry, because both of its subjects ceased to exist in
|
`ENV PATH=/home/builder/go/bin:$PATH` is gone with the `go install`
|
||||||
the same change: it compared a linter binary against `GOLANGCI_LINT_VERSION`
|
that justified it. `script/verify-linter-pin` is retired, deleted
|
||||||
in `script/bootstrap`, and there is now neither a binary crossing between
|
along with its README entry, because both of its subjects ceased to
|
||||||
stages nor a version pin in bootstrap. The drift it guarded has not gone away,
|
exist in the same change: it compared a linter binary against
|
||||||
it has moved — the linter is still pinned twice, now as the `FROM` line of
|
`GOLANGCI_LINT_VERSION` in `script/bootstrap`, and there is now
|
||||||
`Dockerfile.lint` and the `FROM` line of the `Dockerfile` lint stage, with
|
neither a binary crossing between stages nor a version pin in
|
||||||
nothing syncing them, which is exactly what
|
bootstrap. The drift it guarded has not gone away, it has moved — the
|
||||||
|
linter is still pinned twice, now as the `FROM` line of
|
||||||
|
`Dockerfile.lint` and the `FROM` line of the `Dockerfile` lint stage,
|
||||||
|
with nothing syncing them, which is exactly what
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/42 made a build failure. Its
|
https://git.eeqj.de/sneak/sfdupes/issues/42 made a build failure. Its
|
||||||
replacement is one new `script/verify-lint-image-pin`, run as a gate in both
|
replacement is one new `script/verify-lint-image-pin`,
|
||||||
files, which compares the two references to each other and deliberately
|
run as a gate in both files, which compares the two references to
|
||||||
restates neither: a hardcoded expected digest would be a third copy and the
|
each other and deliberately restates neither: a hardcoded expected
|
||||||
same drift one file further out. `golangci-lint config verify` is included per
|
digest would be a third copy and the same drift one file further out.
|
||||||
the ruling, and the concern about its unpinned live HTTPS schema fetch was
|
`golangci-lint config verify` is included per the ruling, and the
|
||||||
measured rather than assumed — under `--network none` the pinned binary both
|
concern about its unpinned live HTTPS schema fetch was measured
|
||||||
passes a valid config and rejects an invalid one with the jsonschema error, so
|
rather than assumed — under `--network none` the pinned binary both
|
||||||
it validates from an embedded schema and makes no network call of its own. The
|
passes a valid config and rejects an invalid one with the jsonschema
|
||||||
README scopes that to the gate steps rather than to linting as a whole:
|
error, so it validates from an embedded schema and makes no network
|
||||||
`Dockerfile.lint` runs `go mod download` above them, so a cold cache still
|
call of its own. The README scopes that to the gate steps rather
|
||||||
needs the network and only a warm one lints offline. Verified: `make lint`
|
than to linting as a whole: `Dockerfile.lint` runs `go mod download`
|
||||||
green with every `PATH` directory containing a `golangci-lint` removed
|
above them, so a cold cache still needs the network and only a warm
|
||||||
|
one lints offline. Verified: `make lint` green with every `PATH`
|
||||||
|
directory containing a `golangci-lint` removed
|
||||||
(`/home/user/go/bin`, `/home/user/.local/bin`, `/usr/local/bin`;
|
(`/home/user/go/bin`, `/home/user/.local/bin`, `/usr/local/bin`;
|
||||||
`command -v golangci-lint` empty); two consecutive `script/lint` runs on an
|
`command -v golangci-lint` empty); two consecutive `script/lint` runs
|
||||||
untouched tree both executed the linter, 27.7s and 28.7s in the lint step
|
on an untouched tree both executed the linter, 27.7s and 28.7s in the
|
||||||
under distinct epochs with the `COPY . .` layer `CACHED` above them, at 42.2s
|
lint step under distinct epochs with the `COPY . .` layer `CACHED`
|
||||||
and 41.8s wall clock — the no-cache rule was not weakened to shorten that.
|
above them, at 42.2s and 41.8s wall clock — the no-cache rule was not
|
||||||
Negative control: a planted `var unusedIssue46Sentinel = 1` failed
|
weakened to shorten that. Negative control: a planted
|
||||||
`script/lint` with
|
`var unusedIssue46Sentinel = 1` failed `script/lint` with
|
||||||
`report.go:173:5: var unusedIssue46Sentinel is unused (unused)`, and failed
|
`report.go:173:5: var unusedIssue46Sentinel is unused (unused)`, and
|
||||||
`make docker` at `[lint 9/9]` with the build stage stopped at `[builder 3/12]`
|
failed `make docker` at `[lint 9/9]` with the build stage stopped at
|
||||||
— `COPY --from=lint`, `script/bootstrap`, the test gate and `make build` all
|
`[builder 3/12]` — `COPY --from=lint`, `script/bootstrap`, the test
|
||||||
zero occurrences — then reverted clean. The drift guard fails on a tag-only
|
gate and `make build` all zero occurrences — then reverted clean. The
|
||||||
disagreement, on a digest-only disagreement, and on an unreadable reference,
|
drift guard fails on a tag-only disagreement, on a digest-only
|
||||||
naming both sides. `make docker` green in 5m35s with all six gates executing
|
disagreement, and on an unreadable reference, naming both sides.
|
||||||
under one epoch (lint 37.6s, test 25.2s reporting
|
`make docker` green in 5m35s with all six gates executing under one
|
||||||
`ok sneak.berlin/go/sfdupes 1.938s coverage: 88.5%`, not `(cached)`). The
|
epoch (lint 37.6s, test 25.2s reporting
|
||||||
non-root quirk still holds: in the builder image with the Go test cache off,
|
`ok sneak.berlin/go/sfdupes 1.938s coverage: 88.5%`, not `(cached)`).
|
||||||
`--user 0:0` fails `TestScanHardlinkRunFailsTogether` (exit 1) where the
|
The non-root quirk still holds: in the builder image with the Go test
|
||||||
unprivileged user passes (exit 0). Noted for follow-up, not fixed here:
|
cache off, `--user 0:0` fails `TestScanHardlinkRunFailsTogether`
|
||||||
`golangci-lint` warns that the `gomodguard` linter is deprecated since v2.12.0
|
(exit 1) where the unprivileged user passes (exit 0). Noted for
|
||||||
in favour of `gomodguard_v2`.
|
follow-up, not fixed here: `golangci-lint` warns that the
|
||||||
|
`gomodguard` linter is deprecated since v2.12.0 in favour of
|
||||||
|
`gomodguard_v2`.
|
||||||
|
|
||||||
- install the Docker build stage's prerequisites by running `script/bootstrap`
|
- install the Docker build stage's prerequisites by running
|
||||||
instead of `apk add --no-cache make` inline (2026-08-09, branch
|
`script/bootstrap` instead of `apk add --no-cache make` inline
|
||||||
`dockerfile-bootstrap`, closes #42): canonical `REPO_POLICIES.md:97` requires
|
(2026-08-09, branch `dockerfile-bootstrap`, closes #42): canonical
|
||||||
it, and the inline install left the build stage maintaining its own notion of
|
`REPO_POLICIES.md:97` requires it, and the inline install left the
|
||||||
the toolchain — exactly the divergence #24 exists to close, one layer down.
|
build stage maintaining its own notion of the toolchain — exactly
|
||||||
The stage now copies `script/` plus `go.mod`/`go.sum` and runs
|
the divergence #24 exists to close, one layer down. The stage now
|
||||||
`script/bootstrap`, which ends in `go mod download`, so the separate
|
copies `script/` plus `go.mod`/`go.sum` and runs `script/bootstrap`,
|
||||||
invocation of that is gone. `COPY --from=lint /usr/bin/golangci-lint` stays,
|
which ends in `go mod download`, so the separate invocation of that
|
||||||
and moves above the bootstrap layer. It is the only edge making this stage
|
is gone. `COPY --from=lint /usr/bin/golangci-lint` stays, and moves
|
||||||
depend on the lint stage, so deleting it as redundant would end fail-fast
|
above the bootstrap layer. It is the only edge making this stage
|
||||||
linting silently. Letting bootstrap install its own linter here would have
|
depend on the lint stage, so deleting it as redundant would end
|
||||||
reintroduced the second toolchain and paid for a from-source build of it. What
|
fail-fast linting silently. Letting bootstrap install its own linter
|
||||||
makes the two stages provably one toolchain rather than two that happen to
|
here would have reintroduced the second toolchain and paid for a
|
||||||
agree is a new `script/verify-linter-pin`, run in the build stage on the
|
from-source build of it. What makes the two stages provably one
|
||||||
binary that arrives from the lint stage, before bootstrap: it fails the build
|
toolchain rather than two that happen to agree is a new
|
||||||
naming both versions unless that binary is the version `script/bootstrap`
|
`script/verify-linter-pin`, run in the build stage on the binary
|
||||||
pins. Bootstrap's own check could not serve that purpose — it reinstalls its
|
that arrives from the lint stage, before bootstrap: it fails the
|
||||||
pin from source and then verifies whatever `PATH` resolves, so drift
|
build naming both versions unless that binary is the version
|
||||||
self-heals silently and a lint stage image bumped on its own would lint at the
|
`script/bootstrap` pins. Bootstrap's own check could not serve that
|
||||||
new version while `make check` ran at the old one, green. The linter version
|
purpose — it reinstalls its pin from source and then verifies
|
||||||
is pinned in two independent places (the lint stage image digest and
|
whatever `PATH` resolves, so drift self-heals silently and a lint
|
||||||
|
stage image bumped on its own would lint at the new version while
|
||||||
|
`make check` ran at the old one, green. The linter version is pinned
|
||||||
|
in two independent places (the lint stage image digest and
|
||||||
`GOLANGCI_LINT_VERSION`) and nothing else keeps them in sync, so a
|
`GOLANGCI_LINT_VERSION`) and nothing else keeps them in sync, so a
|
||||||
half-applied bump is now a build failure. The pin is read out of
|
half-applied bump is now a build failure. The pin is read out of
|
||||||
`script/bootstrap`, which stays the single source of truth; a pin that cannot
|
`script/bootstrap`, which stays the single source of truth; a pin
|
||||||
be read is a hard failure, not a skip. The check needs no `CHECK_EPOCH`: its
|
that cannot be read is a hard failure, not a skip. The check needs
|
||||||
only inputs are the copied binary and `script/`, so Docker invalidates the
|
no `CHECK_EPOCH`: its only inputs are the copied binary and
|
||||||
layer exactly when a cached result would stop being true, and it is documented
|
`script/`, so Docker invalidates the layer exactly when a cached
|
||||||
with the other entrypoints in the README. `$GOPATH/bin` joins `PATH` because
|
result would stop being true, and it is documented with the other
|
||||||
that is where bootstrap's `go install` lands and bootstrap verifies its
|
entrypoints in the README. `$GOPATH/bin` joins `PATH` because
|
||||||
installs against what `PATH` resolves — nothing in the image is shadowed by
|
that is where bootstrap's `go install` lands and bootstrap verifies
|
||||||
it, the directory does not exist until bootstrap runs. Everything added sits
|
its installs against what `PATH` resolves — nothing in the image is
|
||||||
above `ARG CHECK_EPOCH`, and the `chown` and `USER builder` still precede
|
shadowed by it, the directory does not exist until bootstrap runs.
|
||||||
`make check`. Verified: the guard fails the build with both versions named
|
Everything added sits above `ARG CHECK_EPOCH`, and the `chown` and
|
||||||
when the lint stage's linter is faked to a different version, and an
|
`USER builder` still precede `make check`. Verified: the guard fails
|
||||||
unmodified build still passes it; bootstrap runs clean under Alpine's `sh` and
|
the build with both versions named when the lint stage's linter is
|
||||||
its `apk` branch, installing `git` and `make` and finding the copied linter
|
faked to a different version, and an unmodified build still passes
|
||||||
already at the pin; a second build served the bootstrap and dependency layers
|
it; bootstrap runs clean under Alpine's `sh` and its `apk` branch,
|
||||||
`CACHED` while both gates ran with a fresh epoch; a planted `unused` finding
|
installing `git` and `make` and finding the copied
|
||||||
failed the build at the lint gate in 48.9s with the build stage's `make check`
|
linter already at the pin; a second build served the bootstrap and
|
||||||
never starting; and the suite run in the image as `--user 0:0` fails
|
dependency layers `CACHED` while both gates ran with a fresh epoch;
|
||||||
`TestScanHardlinkRunFailsTogether`, so the drop to the unprivileged user is
|
a planted `unused` finding failed the build at the lint gate in
|
||||||
still load-bearing. That last check needs the Go test cache disabled — the
|
48.9s with the build stage's `make check` never starting; and the
|
||||||
first attempt reported `ok ... (cached)` as root, reusing the result the
|
suite run in the image as `--user 0:0` fails
|
||||||
build-time run had left in the shared cache, which would have read as a pass.
|
`TestScanHardlinkRunFailsTogether`, so the drop to the unprivileged
|
||||||
Build wall time, on a shared host running many concurrent builds and so noisy:
|
user is still load-bearing. That last check needs the Go test cache
|
||||||
2m13s on an unchanged tree, 2m17s and 4m29s for two builds after a source
|
disabled — the first attempt reported `ok ... (cached)` as root,
|
||||||
change, 5m14s cold. Only the cold one breaches the policy ceiling, and not
|
reusing the result the build-time run had left in the shared cache,
|
||||||
because of this change — `chown -R builder:builder /src /home/builder` walks
|
which would have read as a pass. Build wall time, on a shared host
|
||||||
the module cache and re-runs on every source change, and it alone varied
|
running many concurrent builds and so noisy: 2m13s on an unchanged
|
||||||
between 77s and 210s across those four builds, which is also the whole spread
|
tree, 2m17s and 4m29s for two builds after a source change, 5m14s
|
||||||
in the totals. The same cold measurement against `main` is 5m03s with a 209s
|
cold. Only the cold one breaches the policy ceiling, and not because
|
||||||
`chown`. Filed as #43
|
of this change — `chown -R builder:builder /src /home/builder` walks
|
||||||
- bust the Docker layer cache for the gate steps, so `script/cibuild` and
|
the module cache and re-runs on every source change, and it alone
|
||||||
`script/docker` cannot report a green they did not earn (2026-08-09, branch
|
varied between 77s and 210s across those four builds, which is also
|
||||||
`cibuild-cache-bust`, closes #32): both scripts were bare `docker build`
|
the whole spread in the totals. The same cold measurement against
|
||||||
invocations with no cache control, and the `Dockerfile` copies the tree before
|
`main` is 5m03s with a 209s `chown`. Filed as #43
|
||||||
running its gates, so on an unchanged tree Docker served those layers from
|
- bust the Docker layer cache for the gate steps, so `script/cibuild`
|
||||||
cache and the build exited 0 having executed nothing. That is not hypothetical
|
and `script/docker` cannot report a green they did not earn
|
||||||
here — every merge this repo has done is a non-fast-forward merge of an
|
(2026-08-09, branch `cibuild-cache-bust`, closes #32): both scripts
|
||||||
undiverged branch, so each merge commit's tree is byte-identical to the branch
|
were bare `docker build` invocations with no cache control, and the
|
||||||
head's and each merge CI run was almost certainly a full cache hit; and PR
|
`Dockerfile` copies the tree before running its gates, so on an
|
||||||
#31's reviewer found `make docker` returning success as a 17-layer cache hit,
|
unchanged tree Docker served those layers from cache and the build
|
||||||
catching it only by being suspicious. The fix is `ARG CHECK_EPOCH` with the
|
exited 0 having executed nothing. That is not hypothetical here —
|
||||||
scripts passing `--build-arg CHECK_EPOCH="$(date +%s)"`. Two details make or
|
every merge this repo has done is a non-fast-forward merge of an
|
||||||
break it. `ARG` is scoped per stage and this `Dockerfile` has three gates
|
undiverged branch, so each merge commit's tree is byte-identical to
|
||||||
across two — `make fmt-check` and `make lint` in the lint stage, `make check`
|
the branch head's and each merge CI run was almost certainly a full
|
||||||
in the build stage — so a single declaration would have left one stage
|
cache hit; and PR #31's reviewer found `make docker` returning
|
||||||
silently cacheable; it is declared in both. And BuildKit hashes the expanded
|
success as a 17-layer cache hit, catching it only by being
|
||||||
command, not the declaration, so a declared-but-unreferenced `ARG` invalidates
|
suspicious. The fix is `ARG CHECK_EPOCH` with the scripts passing
|
||||||
nothing: each gate `RUN` echoes the epoch, which also puts the value in the
|
`--build-arg CHECK_EPOCH="$(date +%s)"`. Two details make or break
|
||||||
build log as evidence the layer really ran. Placement is below the dependency
|
it. `ARG` is scoped per stage and this `Dockerfile` has three gates
|
||||||
layers on purpose — a build that goes cold every time would be a different
|
across two — `make fmt-check` and `make lint` in the lint stage,
|
||||||
bug, not a fix. Verified by running each script twice back to back on an
|
`make check` in the build stage — so a single declaration would have
|
||||||
unchanged tree under `BUILDKIT_PROGRESS=plain`: all three gates executed on
|
left one stage silently cacheable; it is declared in both. And
|
||||||
all four runs, each with a fresh epoch in the log (`script/cibuild` 78.8s then
|
BuildKit hashes the expanded command, not the declaration, so a
|
||||||
61.1s; `script/docker` 61.1s then 53.4s), and twelve steps were still served
|
declared-but-unreferenced `ARG` invalidates nothing: each gate `RUN`
|
||||||
`CACHED` in the steady state — both `go mod download`s, `apk add`, `adduser`,
|
echoes the epoch, which also puts the value in the build log as
|
||||||
the `chown`, every `go.mod`/`go.sum` and source copy, the linter copy out of
|
evidence the layer really ran. Placement is below the dependency
|
||||||
the lint stage, and the binary copy into the runtime stage. The lint stage
|
layers on purpose — a build that goes cold every time would be a
|
||||||
still gates the build stage: with a deliberate `unused` finding planted in the
|
different bug, not a fix. Verified by running each script twice back
|
||||||
tree, the build failed at `make lint` in 36.1s and the build-stage
|
to back on an unchanged tree under `BUILDKIT_PROGRESS=plain`: all
|
||||||
`make check` never started. The build stage also still drops to the
|
three gates executed on all four runs, each with a fresh epoch in
|
||||||
unprivileged `builder` user before `make check`, which the suite depends on
|
the log (`script/cibuild` 78.8s then 61.1s; `script/docker` 61.1s
|
||||||
rather than merely prefers: forcing the same image to run the tests as root
|
then 53.4s), and twelve steps were still served `CACHED` in the
|
||||||
fails `TestScanHardlinkRunFailsTogether`, because root reads straight through
|
steady state — both `go mod download`s, `apk add`, `adduser`, the
|
||||||
the `chmod(0)` the test uses to prove hard links are read once. This is the
|
`chown`, every `go.mod`/`go.sum` and source copy, the linter copy
|
||||||
local fix only; propagating it to the canonical templates is `prompts` #26
|
out of the lint stage, and the binary copy into the runtime stage.
|
||||||
- check the installed golangci-lint version in `script/bootstrap` instead of
|
The lint stage still gates the build stage: with a deliberate
|
||||||
only its presence (2026-08-09, branch `bootstrap-version-check`, closes #24):
|
`unused` finding planted in the tree, the build failed at
|
||||||
`missing golangci-lint` meant any linter already on `PATH` satisfied the
|
`make lint` in 36.1s and the build-stage `make check` never started.
|
||||||
check, so the pin was never consulted and the v2.12.2 bump from #3 was inert
|
The build stage also still drops to the unprivileged `builder` user
|
||||||
on every host that already had one — this host ran v2.10.1 against a v2.12.2
|
before `make check`, which the suite depends on rather than merely
|
||||||
pin, `make check` went green, and `make docker` then rejected the same commit
|
prefers: forcing the same image to run the tests as root fails
|
||||||
with findings the local gate never saw. The version now lives in one place,
|
`TestScanHardlinkRunFailsTogether`, because root reads straight
|
||||||
`GOLANGCI_LINT_VERSION`, with the `go install` module ref derived from it so a
|
through the `chmod(0)` the test uses to prove hard links are read
|
||||||
bump cannot half-apply; a `golangci_lint_version` helper parses
|
once. This is the local fix only; propagating it to the canonical
|
||||||
`golangci-lint --version` (taking the field after the word `version` and
|
templates is `prompts` #26
|
||||||
tolerating an optional leading `v`, which the module ref carries and the
|
- check the installed golangci-lint version in `script/bootstrap`
|
||||||
binary's output does not), and any version that is not the pin — older, newer,
|
instead of only its presence (2026-08-09, branch
|
||||||
absent or unparseable — is reinstalled. The install is then verified against
|
`bootstrap-version-check`, closes #24): `missing golangci-lint` meant
|
||||||
the binary `PATH` actually resolves: `go install` writes into `GOBIN` (or
|
any linter already on `PATH` satisfied the check, so the pin was never
|
||||||
`GOPATH/bin`) while `make lint` runs whichever `golangci-lint` comes first on
|
consulted and the v2.12.2 bump from #3 was inert on every host that
|
||||||
`PATH`, so a wrong-version one sitting ahead of it — nix, apt, brew, apk, or
|
already had one — this host ran v2.10.1 against a v2.12.2 pin,
|
||||||
the `/usr/local/bin` copy the `Dockerfile` builder stage makes — would swallow
|
`make check` went green, and `make docker` then rejected the same
|
||||||
the install and leave the local gate disagreeing with CI under an affirmative
|
commit with findings the local gate never saw. The version now lives
|
||||||
`bootstrap complete`. Bootstrap now re-reads the effective version after
|
in one place, `GOLANGCI_LINT_VERSION`, with the `go install` module
|
||||||
installing and, on a mismatch, prints both paths and both versions to stderr
|
ref derived from it so a bump cannot half-apply; a
|
||||||
and exits non-zero instead of claiming success; it does not reorder anyone's
|
`golangci_lint_version` helper parses `golangci-lint --version`
|
||||||
|
(taking the field after the word `version` and tolerating an optional
|
||||||
|
leading `v`, which the module ref carries and the binary's output does
|
||||||
|
not), and any version that is not the pin — older, newer, absent or
|
||||||
|
unparseable — is reinstalled. The install is then verified against the
|
||||||
|
binary `PATH` actually resolves: `go install` writes into `GOBIN` (or
|
||||||
|
`GOPATH/bin`) while `make lint` runs whichever `golangci-lint` comes
|
||||||
|
first on `PATH`, so a wrong-version one sitting ahead of it — nix,
|
||||||
|
apt, brew, apk, or the `/usr/local/bin` copy the `Dockerfile` builder
|
||||||
|
stage makes — would swallow the install and leave the local gate
|
||||||
|
disagreeing with CI under an affirmative `bootstrap complete`.
|
||||||
|
Bootstrap now re-reads the effective version after installing and, on
|
||||||
|
a mismatch, prints both paths and both versions to stderr and exits
|
||||||
|
non-zero instead of claiming success; it does not reorder anyone's
|
||||||
`PATH` or delete their binary. The `--version` call keeps its stderr
|
`PATH` or delete their binary. The `--version` call keeps its stderr
|
||||||
connected, so a present-but-broken binary says why rather than reinstalling
|
connected, so a present-but-broken binary says why rather than
|
||||||
forever in silence, and is bounded by `timeout(1)` where that exists, so a
|
reinstalling forever in silence, and is bounded by `timeout(1)` where
|
||||||
wedged binary cannot hang bootstrap. `git`, `make` and `go` keep their
|
that exists, so a wedged binary cannot hang bootstrap. `git`, `make`
|
||||||
presence-only checks and now say why in a comment: they are host
|
and `go` keep their presence-only checks and now say why in a
|
||||||
package-manager tools the repo deliberately does not pin, with `go.mod`
|
comment: they are host package-manager tools the repo deliberately
|
||||||
governing the language version and the digest-pinned images covering
|
does not pin, with `go.mod` governing the language version and the
|
||||||
reproducible builds. Verified on this host by bootstrapping from v2.10.1 to
|
digest-pinned images covering reproducible builds. Verified on this
|
||||||
v2.12.2 and running it again to a no-op, plus stub runs of the real script
|
host by bootstrapping from v2.10.1 to v2.12.2 and running it again to
|
||||||
under `dash` covering a thirteen-input parse matrix (absent, older, newer,
|
a no-op, plus stub runs of the real script under `dash` covering a
|
||||||
host-style, image-style, leading-`v`, stderr-only, empty, non-zero exit,
|
thirteen-input parse matrix (absent, older, newer, host-style,
|
||||||
impostor binary, `(devel)`, trailing `version`), a shadowed install that must
|
image-style, leading-`v`, stderr-only, empty, non-zero exit, impostor
|
||||||
exit non-zero, an install destination not on `PATH` at all, `GOBIN` set, and a
|
binary, `(devel)`, trailing `version`), a shadowed install that must
|
||||||
wedged binary that must hit the timeout; `make check` and `make lint` are
|
exit non-zero, an install destination not on `PATH` at all, `GOBIN`
|
||||||
clean at v2.12.2, so v2.10.1 was not hiding any findings on `main`
|
set, and a wedged binary that must hit the timeout; `make check` and
|
||||||
|
`make lint` are clean at v2.12.2, so v2.10.1 was not hiding any
|
||||||
|
findings on `main`
|
||||||
- unwind the hash worker pool on the error path (2026-08-09, branch
|
- unwind the hash worker pool on the error path (2026-08-09, branch
|
||||||
`hash-pool-cleanup`, closes #6): `hashPhase` used to return the moment
|
`hash-pool-cleanup`, closes #6): `hashPhase` used to return the
|
||||||
`recordRun` failed and abandon the pool — the feeder parked forever on a full
|
moment `recordRun` failed and abandon the pool — the feeder parked
|
||||||
`jobs` channel and every worker on a full `results` channel. That only stopped
|
forever on a full `jobs` channel and every worker on a full
|
||||||
being invisible when #4 landed and `runScan` began unwinding instead of
|
`results` channel. That only stopped being invisible when #4 landed
|
||||||
calling `os.Exit`. The pool is now an owned, context-aware `hashPool`: every
|
and `runScan` began unwinding instead of calling `os.Exit`. The
|
||||||
blocking send in the feeder and the workers selects on `ctx.Done()`, `jobs` is
|
pool is now an owned, context-aware `hashPool`: every blocking send
|
||||||
closed on every path out, and `hashPhase` defers `pool.stop()`, which cancels
|
in the feeder and the workers selects on `ctx.Done()`, `jobs` is
|
||||||
and then drains `results` until the last goroutine has exited — draining is
|
closed on every path out, and `hashPhase` defers `pool.stop()`,
|
||||||
what frees a worker already parked on a send. `ctx` is threaded from
|
which cancels and then drains `results` until the last goroutine
|
||||||
`cmd.Context()` through `runScan`, `syncScan`, both worker pools and the whole
|
has exited — draining is what frees a worker already parked on a
|
||||||
database layer (it is the first parameter everywhere), so #5 can hand this
|
send. `ctx` is threaded from `cmd.Context()` through `runScan`,
|
||||||
path a signal and needs to add nothing else. The walk pool never leaked,
|
`syncScan`, both worker pools and the whole database layer (it is
|
||||||
because `walkPhase` always drains its events to close, but it has the same
|
the first parameter everywhere), so #5 can hand this path a signal
|
||||||
unbounded-send shape and #5 will give it an early return, so it gets the same
|
and needs to add nothing else. The walk pool never leaked, because
|
||||||
treatment plus a `ctx.Err()` guard after the walk: a cancelled walk yields a
|
`walkPhase` always drains its events to close, but it has the same
|
||||||
partial size census, and every file it never reached looks vanished to the
|
unbounded-send shape and #5 will give it an early return, so it
|
||||||
update phase. That phase's own `BeginTx` fails on the same cancelled context
|
gets the same treatment plus a `ctx.Err()` guard after the walk: a
|
||||||
before deleting anything, so the guard is defence in depth rather than the
|
cancelled walk yields a partial size census, and every file it never
|
||||||
only barrier — but it is the one that survives #5 deciding an interrupted scan
|
reached looks vanished to the update phase. That phase's own
|
||||||
may commit what it has. Tests drive `run(scan)` against a database whose
|
`BeginTx` fails on the same cancelled context before deleting
|
||||||
insert trigger aborts, and assert both that the scan fails instead of hanging
|
anything, so the guard is defence in depth rather than the only
|
||||||
and that `runtime.NumGoroutine()` polls back to its pre-scan baseline; a
|
barrier — but it is the one that survives #5 deciding an interrupted
|
||||||
second set cancels a scan part-way through the walk — deterministically, by
|
scan may commit what it has. Tests drive `run(scan)` against a
|
||||||
counting the scan's own consultations of `ctx.Done()` rather than racing a
|
database whose insert trigger aborts, and assert both that the scan
|
||||||
timer — and asserts that it stops at the guard holding a partial census and a
|
fails instead of hanging and that `runtime.NumGoroutine()` polls
|
||||||
still-populated record index, with every record intact. The remaining
|
back to its pre-scan baseline; a second set cancels a scan part-way
|
||||||
cancellation branches of both pools are covered by direct tests of
|
through the walk — deterministically, by counting the scan's own
|
||||||
`sendEvent`, the walk workers, `dispatchDirs`, `feedHashJobs`, `hashWorker`
|
consultations of `ctx.Done()` rather than racing a timer — and
|
||||||
and `hashPhase`
|
asserts that it stops at the guard holding a partial census and a
|
||||||
- guarantee the database is closed on every fatal exit path (2026-08-09, branch
|
still-populated record index, with every record intact. The
|
||||||
`db-close-on-fatal`, closes #4): `fatalf` and its `os.Exit(1)` are gone, so
|
remaining cancellation branches of both pools are covered by direct
|
||||||
the deferred `db.Close()` — and with it the SQLite WAL checkpoint — now
|
tests of `sendEvent`, the walk workers, `dispatchDirs`,
|
||||||
actually runs when a subcommand fails; `runScan`, `runReport`, `runTrees`,
|
`feedHashJobs`, `hashWorker` and `hashPhase`
|
||||||
`loadRecords` and `resolveRoots` return errors instead. The single exit point
|
- guarantee the database is closed on every fatal exit path
|
||||||
is `run` in `main.go`: it maps a `fatalError` (anything a subcommand returned)
|
(2026-08-09, branch `db-close-on-fatal`, closes #4): `fatalf` and
|
||||||
to exit 1 and cobra's own argument and flag errors to exit 2, which keeps a
|
its `os.Exit(1)` are gone, so the deferred `db.Close()` — and with
|
||||||
runtime failure from being reported as a usage error or printing the usage
|
it the SQLite WAL checkpoint — now actually runs when a subcommand
|
||||||
text. New `main_test.go` drives the CLI in-process and asserts the exit codes
|
fails; `runScan`, `runReport`, `runTrees`, `loadRecords` and
|
||||||
from README §Error handling plus the stdout/stderr split, including that a
|
`resolveRoots` return errors instead. The single exit point is `run`
|
||||||
fatal error raised after the database is open leaves no `-wal`/`-shm` sidecar
|
in `main.go`: it maps a `fatalError` (anything a subcommand
|
||||||
behind for `scan`, `report` or `trees`
|
returned) to exit 1 and cobra's own argument and flag errors to exit
|
||||||
- update golangci-lint to v2.12.2 with the canonical config (2026-08-09, branch
|
2, which keeps a runtime failure from being reported as a usage
|
||||||
`golangci-v2.12.2`, merged as `38a01bd`, closes #3): bumped the pinned linter
|
error or printing the usage text. New `main_test.go` drives the CLI
|
||||||
in the `Dockerfile` lint stage and `script/bootstrap` from v2.12.1 to v2.12.2,
|
in-process and asserts the exit codes from README §Error handling
|
||||||
and replaced `.golangci.yml` with the canonical file — the linter settings
|
plus the stdout/stderr split, including that a fatal error raised
|
||||||
|
after the database is open leaves no `-wal`/`-shm` sidecar behind
|
||||||
|
for `scan`, `report` or `trees`
|
||||||
|
- update golangci-lint to v2.12.2 with the canonical config
|
||||||
|
(2026-08-09, branch `golangci-v2.12.2`, merged as `38a01bd`,
|
||||||
|
closes #3): bumped the pinned linter in the `Dockerfile` lint
|
||||||
|
stage and `script/bootstrap` from v2.12.1 to v2.12.2, and replaced
|
||||||
|
`.golangci.yml` with the canonical file — the linter settings
|
||||||
(`lll`, `funlen`, `cyclop`, `dupl` thresholds) now live under
|
(`lll`, `funlen`, `cyclop`, `dupl` thresholds) now live under
|
||||||
`linters.settings` per the v2 schema, so they are actually applied; no new
|
`linters.settings` per the v2 schema, so they are actually
|
||||||
lint findings surfaced
|
applied; no new lint findings surfaced
|
||||||
- convert Makefile targets to scripts-to-rule-them-all `script/` entrypoints
|
- convert Makefile targets to scripts-to-rule-them-all `script/`
|
||||||
like the other managed repos (2026-07-26, commit `3abeacf`, closes #1): all 12
|
entrypoints like the other managed repos (2026-07-26, commit
|
||||||
`script/` entrypoints exist (`bootstrap`, `setup`, `projectname`, `test`,
|
`3abeacf`, closes #1): all 12 `script/` entrypoints exist
|
||||||
`lint`, `fmt`, `fmt-check`, `check`, `docker`, `cibuild`, `precommit`,
|
(`bootstrap`, `setup`, `projectname`, `test`, `lint`, `fmt`,
|
||||||
`install-precommit`) and every Makefile target is now a thin shim over them,
|
`fmt-check`, `check`, `docker`, `cibuild`, `precommit`,
|
||||||
matching the other managed repos
|
`install-precommit`) and every Makefile target is now a thin shim
|
||||||
|
over them, matching the other managed repos
|
||||||
- make the binary the default Make target (2026-07-24, branch
|
- make the binary the default Make target (2026-07-24, branch
|
||||||
`make-default-target`): plain `make` now builds `sfdupes` (previously it ran
|
`make-default-target`): plain `make` now builds `sfdupes`
|
||||||
`check` plus `build`); `make build` remains as an alias
|
(previously it ran `check` plus `build`); `make build` remains as
|
||||||
- scan-wide phases, concurrent operands, batched updates (2026-07-24, branch
|
an alias
|
||||||
`scan-wide-phases`): all operands seed the shared walk pool and every pass
|
- scan-wide phases, concurrent operands, batched updates (2026-07-24,
|
||||||
runs once over the whole scan, so totals and ETAs are scan-global; the
|
branch `scan-wide-phases`): all operands seed the shared walk pool
|
||||||
per-operand walk/hash/update cycles and their stderr announcements are gone;
|
and every pass runs once over the whole scan, so totals and ETAs
|
||||||
the update pass commits in batched transactions — the filesystem is
|
are scan-global; the per-operand walk/hash/update cycles and their
|
||||||
authoritative and the database an eventually-consistent reflection, so
|
stderr announcements are gone; the update pass commits in batched
|
||||||
scan-level atomicity is not required
|
transactions — the filesystem is authoritative and the database an
|
||||||
|
eventually-consistent reflection, so scan-level atomicity is not
|
||||||
|
required
|
||||||
- split the stat pass back out of the walk (2026-07-24, branch
|
- split the stat pass back out of the walk (2026-07-24, branch
|
||||||
`parallel-phases`): phases are strictly sequential again — walk, stat, hash,
|
`parallel-phases`): phases are strictly sequential again — walk,
|
||||||
update per operand — with parallelism only inside each phase; the walk
|
stat, hash, update per operand — with parallelism only inside each
|
||||||
enumerates paths with per-directory workers and the stat pass lstats them with
|
phase; the walk enumerates paths with per-directory workers and the
|
||||||
per-file workers, restoring the exact total/ETA stat bar
|
stat pass lstats them with per-file workers, restoring the exact
|
||||||
- announce each operand on stderr before its passes (2026-07-24, branch
|
total/ETA stat bar
|
||||||
`scan-operand-progress`): with per-operand walk/hash/update cycles, a
|
- announce each operand on stderr before its passes (2026-07-24,
|
||||||
multi-operand run (e.g. `scan /srv/*`) showed pass totals that looked like the
|
branch `scan-operand-progress`): with per-operand walk/hash/update
|
||||||
whole run's — an operator watching operand 3 of 14 hash 300k files concluded
|
cycles, a multi-operand run (e.g. `scan /srv/*`) showed pass totals
|
||||||
20M files were being skipped
|
that looked like the whole run's — an operator watching operand 3 of
|
||||||
- parallel walk (2026-07-24, branch `parallel-walk`): the walk pass was a single
|
14 hash 300k files concluded 20M files were being skipped
|
||||||
goroutine and took hours at ~20M files on a busy pool (observed: 22M files in
|
- parallel walk (2026-07-24, branch `parallel-walk`): the walk pass
|
||||||
4h on a ZFS server); it is now a per-directory worker-pool traversal that
|
was a single goroutine and took hours at ~20M files on a busy pool
|
||||||
records size/mtime during the walk (folding away the separate stat pass,
|
(observed: 22M files in 4h on a ZFS server); it is now a
|
||||||
halving metadata I/O), and each `PATH` operand commits in its own transaction
|
per-directory worker-pool traversal that records size/mtime during
|
||||||
so an interrupted scan keeps completed operands
|
the walk (folding away the separate stat pass, halving metadata
|
||||||
|
I/O), and each `PATH` operand commits in its own transaction so an
|
||||||
|
interrupted scan keeps completed operands
|
||||||
|
|
||||||
- persistent scan database (2026-07-24, branch `persistent-database`): `scan`
|
- persistent scan database (2026-07-24, branch `persistent-database`):
|
||||||
now maintains a SQLite database (`modernc.org/sqlite`, pure Go, cgo stays
|
`scan` now maintains a SQLite database (`modernc.org/sqlite`, pure
|
||||||
disabled) keyed by absolute path that survives between runs — a rescan hashes
|
Go, cgo stays disabled) keyed by absolute path that survives between
|
||||||
only new or changed files (by mtime/size), deletes records for files vanished
|
runs — a rescan hashes only new or changed files (by mtime/size),
|
||||||
from under the scanned operands, and leaves records outside them untouched, so
|
deletes records for files vanished from under the scanned operands,
|
||||||
`scan` can be cronned daily; `report` and `trees` read the database (no
|
and leaves records outside them untouched, so `scan` can be cronned
|
||||||
positional arguments) instead of a scan stream. Database at
|
daily; `report` and `trees` read the database (no positional
|
||||||
`/var/lib/sfdupes/db.sqlite`, overridable via `SFDUPES_DATABASE`; WAL
|
arguments) instead of a scan stream. Database at
|
||||||
journaling plus a single-transaction update keep a report run during a scan
|
`/var/lib/sfdupes/db.sqlite`, overridable via `SFDUPES_DATABASE`;
|
||||||
safe
|
WAL journaling plus a single-transaction update keep a report run
|
||||||
- add the `origin` remote (`git@git.eeqj.de:sneak/sfdupes.git`), tag `v0.0.1`,
|
during a scan safe
|
||||||
and push `main` plus tags (2026-07-23)
|
- add the `origin` remote (`git@git.eeqj.de:sneak/sfdupes.git`), tag
|
||||||
|
`v0.0.1`, and push `main` plus tags (2026-07-23)
|
||||||
- `scan` CLI rework (2026-07-23, branch `scan-required-paths`): required
|
- `scan` CLI rework (2026-07-23, branch `scan-required-paths`): required
|
||||||
`PATH...` operands via cobra flags replacing the `/srv` `-root` default; new
|
`PATH...` operands via cobra flags replacing the `/srv` `-root`
|
||||||
`-x`/`--one-file-system` flag (GNU convention) to stop at filesystem
|
default; new `-x`/`--one-file-system` flag (GNU convention) to stop
|
||||||
boundaries, which are crossed by default
|
at filesystem boundaries, which are crossed by default
|
||||||
- bring the repo into full policy compliance (2026-07-23, branch
|
- bring the repo into full policy compliance (2026-07-23, branch
|
||||||
`repo-policy-compliance`; checklist below)
|
`repo-policy-compliance`; checklist below)
|
||||||
- `git init` with README-only first commit; code baseline committed on `main`
|
- `git init` with README-only first commit; code baseline committed on
|
||||||
(2026-07-22)
|
`main` (2026-07-22)
|
||||||
- implement `scan`, `report`, and `trees` subcommands (pre-git history)
|
- implement `scan`, `report`, and `trees` subcommands (pre-git history)
|
||||||
|
|
||||||
# Future Steps
|
# Future Steps
|
||||||
|
|
||||||
- possible later features (explicitly out of scope per README): full-content
|
- possible later features (explicitly out of scope per README):
|
||||||
verification of candidates, removal-script helpers
|
full-content verification of candidates, removal-script helpers
|
||||||
|
|
||||||
# Repo Policy Compliance
|
# Repo Policy Compliance
|
||||||
|
|
||||||
Audited 2026-07-22 against `REPO_POLICIES.md` (2026-07-06), the existing repo
|
Audited 2026-07-22 against `REPO_POLICIES.md` (2026-07-06), the existing
|
||||||
checklist, and the Go styleguide. Code is already gofmt-clean, so no standalone
|
repo checklist, and the Go styleguide. Code is already gofmt-clean, so no
|
||||||
formatting commit is needed.
|
standalone formatting commit is needed.
|
||||||
|
|
||||||
- [x] `.gitignore` missing — the compiled `sfdupes` binary and `files.dat` sit
|
- [x] `.gitignore` missing — the compiled `sfdupes` binary and
|
||||||
untracked in the tree; needs OS/editor/Go artifacts plus secrets patterns
|
`files.dat` sit untracked in the tree; needs OS/editor/Go
|
||||||
|
artifacts plus secrets patterns
|
||||||
- [x] `.editorconfig` missing
|
- [x] `.editorconfig` missing
|
||||||
- [x] `LICENSE` missing and README has no License section (MIT assumed from
|
- [x] `LICENSE` missing and README has no License section (MIT assumed
|
||||||
house convention — user to confirm)
|
from house convention — user to confirm)
|
||||||
- [x] `REPO_POLICIES.md` missing from repo root
|
- [x] `REPO_POLICIES.md` missing from repo root
|
||||||
- [x] `.golangci.yml` missing (install canonical copy); code must then pass
|
- [x] `.golangci.yml` missing (install canonical copy); code must then
|
||||||
`make lint` (150 findings fixed; `make lint` is clean)
|
pass `make lint` (150 findings fixed; `make lint` is clean)
|
||||||
- [x] `Makefile` lacks required targets `test`, `lint`, `fmt`, `fmt-check`,
|
- [x] `Makefile` lacks required targets `test`, `lint`, `fmt`,
|
||||||
`docker`, `hooks`; `check` currently depends on `build`, which writes the
|
`fmt-check`, `docker`, `hooks`; `check` currently depends on
|
||||||
binary (`make check` must not modify files)
|
`build`, which writes the binary (`make check` must not modify
|
||||||
- [x] no tests — `go test ./...` has nothing to run; policy requires real tests
|
files)
|
||||||
with a 30-second timeout and the conditional `-v` rerun pattern (suite
|
- [x] no tests — `go test ./...` has nothing to run; policy requires
|
||||||
covers parsing, grouping, digests, suppression, hashing, and the scan
|
real tests with a 30-second timeout and the conditional `-v`
|
||||||
pipeline; 64% coverage)
|
rerun pattern (suite covers parsing, grouping, digests,
|
||||||
- [x] `Dockerfile` missing — Go multistage with hash-pinned images: fail-fast
|
suppression, hashing, and the scan pipeline; 64% coverage)
|
||||||
lint stage, build stage running `make check`
|
- [x] `Dockerfile` missing — Go multistage with hash-pinned images:
|
||||||
|
fail-fast lint stage, build stage running `make check`
|
||||||
- [x] `.dockerignore` missing
|
- [x] `.dockerignore` missing
|
||||||
- [x] `.gitea/workflows/check.yml` missing (`docker build .` on push, checkout
|
- [x] `.gitea/workflows/check.yml` missing (`docker build .` on push,
|
||||||
action pinned by commit SHA)
|
checkout action pinned by commit SHA)
|
||||||
- [x] README lacks required sections: Description first line
|
- [x] README lacks required sections: Description first line
|
||||||
(name/purpose/category/license/author), Getting Started, Rationale, TODO,
|
(name/purpose/category/license/author), Getting Started,
|
||||||
License, Author
|
Rationale, TODO, License, Author
|
||||||
- [x] README non-goal "no git repository setup and no CI" is stale now that the
|
- [x] README non-goal "no git repository setup and no CI" is stale now
|
||||||
repo is under git with CI
|
that the repo is under git with CI
|
||||||
- [x] pre-commit hook not installed (`make hooks` once the target exists)
|
- [x] pre-commit hook not installed (`make hooks` once the target
|
||||||
|
exists)
|
||||||
|
|
||||||
Accepted divergences (no action):
|
Accepted divergences (no action):
|
||||||
|
|
||||||
- flat single-package layout with `.go` files in the repo root — fine for a
|
- flat single-package layout with `.go` files in the repo root — fine
|
||||||
small single-binary tool per the Go styleguide; the tracker audit agrees
|
for a small single-binary tool per the Go styleguide; the tracker
|
||||||
- `go test` runs without `-race` — the repo mandates `CGO_ENABLED=0` (pure-Go
|
audit agrees
|
||||||
builds) and the race detector requires cgo
|
- `go test` runs without `-race` — the repo mandates `CGO_ENABLED=0`
|
||||||
|
(pure-Go builds) and the race detector requires cgo
|
||||||
|
|||||||
+1
-1
@@ -456,7 +456,7 @@ func TestHashWorkerDropsQueuedRuns(t *testing.T) {
|
|||||||
go func() {
|
go func() {
|
||||||
defer close(done)
|
defer close(done)
|
||||||
|
|
||||||
hashWorker(cancelledContext(t), jobs, results)
|
hashWorker(cancelledContext(t), jobs, results, hashSignature)
|
||||||
}()
|
}()
|
||||||
|
|
||||||
awaitReturn(t, done, "hashWorker")
|
awaitReturn(t, done, "hashWorker")
|
||||||
|
|||||||
@@ -40,18 +40,20 @@ CREATE TABLE files (
|
|||||||
size INTEGER NOT NULL,
|
size INTEGER NOT NULL,
|
||||||
mtime INTEGER NOT NULL,
|
mtime INTEGER NOT NULL,
|
||||||
head TEXT NOT NULL,
|
head TEXT NOT NULL,
|
||||||
tail TEXT NOT NULL
|
tail TEXT NOT NULL,
|
||||||
|
content TEXT NOT NULL
|
||||||
) WITHOUT ROWID
|
) WITHOUT ROWID
|
||||||
`
|
`
|
||||||
|
|
||||||
// upsertSQL inserts one file record, replacing any existing record for
|
// upsertSQL inserts one file record, replacing any existing record for
|
||||||
// the same path.
|
// the same path.
|
||||||
const upsertSQL = `
|
const upsertSQL = `
|
||||||
INSERT INTO files (path, size, mtime, head, tail)
|
INSERT INTO files (path, size, mtime, head, tail, content)
|
||||||
VALUES (?, ?, ?, ?, ?)
|
VALUES (?, ?, ?, ?, ?, ?)
|
||||||
ON CONFLICT (path) DO UPDATE SET
|
ON CONFLICT (path) DO UPDATE SET
|
||||||
size = excluded.size, mtime = excluded.mtime,
|
size = excluded.size, mtime = excluded.mtime,
|
||||||
head = excluded.head, tail = excluded.tail
|
head = excluded.head, tail = excluded.tail,
|
||||||
|
content = excluded.content
|
||||||
`
|
`
|
||||||
|
|
||||||
// errNoDatabase reports a missing database file for report/trees.
|
// errNoDatabase reports a missing database file for report/trees.
|
||||||
@@ -205,7 +207,7 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) {
|
|||||||
// loadFileRows reads every record from the files table.
|
// loadFileRows reads every record from the files table.
|
||||||
func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
||||||
rows, err := db.QueryContext(ctx,
|
rows, err := db.QueryContext(ctx,
|
||||||
"SELECT path, size, mtime, head, tail FROM files")
|
"SELECT path, size, mtime, head, tail, content FROM files")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("read records: %w", err)
|
return nil, fmt.Errorf("read records: %w", err)
|
||||||
}
|
}
|
||||||
@@ -220,7 +222,8 @@ func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
|||||||
r scanRec
|
r scanRec
|
||||||
)
|
)
|
||||||
|
|
||||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail)
|
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
|
||||||
|
&r.content)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("read record: %w", err)
|
return nil, fmt.Errorf("read record: %w", err)
|
||||||
}
|
}
|
||||||
@@ -275,6 +278,62 @@ func loadFileMeta(ctx context.Context, db *sql.DB,
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// contentCandidatesSQL selects every record of at least headTailMin
|
||||||
|
// bytes whose size, head, and tail equal another record's, in each
|
||||||
|
// group (the records sharing a size, head, and tail) where at least one
|
||||||
|
// record has no content hash, with whether each record has one. SQLite
|
||||||
|
// does the grouping, so no other record's hashes are loaded into
|
||||||
|
// memory; the rows come ordered by size, head, and tail, so each
|
||||||
|
// group's rows arrive together.
|
||||||
|
const contentCandidatesSQL = `
|
||||||
|
SELECT f.path, f.size, f.mtime, f.head, f.tail, f.content <> ''
|
||||||
|
FROM files AS f
|
||||||
|
JOIN (
|
||||||
|
SELECT size, head, tail
|
||||||
|
FROM files
|
||||||
|
WHERE size >= ? AND head <> ''
|
||||||
|
GROUP BY size, head, tail
|
||||||
|
HAVING COUNT(*) > 1 AND SUM(content = '') > 0
|
||||||
|
) AS g USING (size, head, tail)
|
||||||
|
ORDER BY size, head, tail
|
||||||
|
`
|
||||||
|
|
||||||
|
// loadContentCandidates streams the rows of contentCandidatesSQL to fn:
|
||||||
|
// each record, without its content hash, and whether it has one.
|
||||||
|
func loadContentCandidates(ctx context.Context, db *sql.DB,
|
||||||
|
fn func(r scanRec, hashed bool),
|
||||||
|
) error {
|
||||||
|
rows, err := db.QueryContext(ctx, contentCandidatesSQL, headTailMin)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("read records: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
defer func() { _ = rows.Close() }()
|
||||||
|
|
||||||
|
for rows.Next() {
|
||||||
|
var (
|
||||||
|
path []byte
|
||||||
|
r scanRec
|
||||||
|
hashed int64
|
||||||
|
)
|
||||||
|
|
||||||
|
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, &hashed)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("read record: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
r.path = string(path)
|
||||||
|
fn(r, hashed != 0)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = rows.Err()
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("read records: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
// updateBatchSize is the number of record changes committed per
|
// updateBatchSize is the number of record changes committed per
|
||||||
// transaction during the update pass. The filesystem is authoritative
|
// transaction during the update pass. The filesystem is authoritative
|
||||||
// and the database an eventually-consistent reflection of it, so
|
// and the database an eventually-consistent reflection of it, so
|
||||||
@@ -348,7 +407,7 @@ func execUpserts(ctx context.Context, tx *sql.Tx, upserts []scanRec,
|
|||||||
|
|
||||||
for _, r := range upserts {
|
for _, r := range upserts {
|
||||||
_, err = st.ExecContext(ctx,
|
_, err = st.ExecContext(ctx,
|
||||||
[]byte(r.path), r.size, r.mtime, r.head, r.tail)
|
[]byte(r.path), r.size, r.mtime, r.head, r.tail, r.content)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("upsert %s: %w", r.path, err)
|
return fmt.Errorf("upsert %s: %w", r.path, err)
|
||||||
}
|
}
|
||||||
|
|||||||
+10
-4
@@ -136,10 +136,14 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
|||||||
db := openTestDB(t)
|
db := openTestDB(t)
|
||||||
|
|
||||||
// Paths may contain tabs and newlines; the database must store
|
// Paths may contain tabs and newlines; the database must store
|
||||||
// them byte-exactly.
|
// them byte-exactly. Every hash, content included, comes back as
|
||||||
|
// written.
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 2, mtime: 20, head: "h2", tail: "t2", path: "/a/tab\tnew\nline"},
|
{
|
||||||
{size: 1, mtime: 10, head: "h1", tail: "t1", path: "/a/x"},
|
size: 2, mtime: 20, head: "h2", tail: "t2", content: "c2",
|
||||||
|
path: "/a/tab\tnew\nline",
|
||||||
|
},
|
||||||
|
{size: 1, mtime: 10, head: "h1", tail: "t1", content: "c1", path: "/a/x"},
|
||||||
}
|
}
|
||||||
|
|
||||||
err := applyChanges(t.Context(), db, recs, nil,
|
err := applyChanges(t.Context(), db, recs, nil,
|
||||||
@@ -163,7 +167,9 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
|||||||
|
|
||||||
// An upsert for an existing path updates in place; a delete
|
// An upsert for an existing path updates in place; a delete
|
||||||
// removes exactly its path.
|
// removes exactly its path.
|
||||||
upd := scanRec{size: 3, mtime: 30, head: "h3", tail: "t3", path: "/a/x"}
|
upd := scanRec{
|
||||||
|
size: 3, mtime: 30, head: "h3", tail: "t3", content: "c3", path: "/a/x",
|
||||||
|
}
|
||||||
|
|
||||||
err = applyChanges(t.Context(), db, []scanRec{upd},
|
err = applyChanges(t.Context(), db, []scanRec{upd},
|
||||||
[]string{"/a/tab\tnew\nline"}, newProgress("update", 2))
|
[]string{"/a/tab\tnew\nline"}, newProgress("update", 2))
|
||||||
|
|||||||
@@ -1,10 +1,14 @@
|
|||||||
// Command sfdupes quickly identifies candidate duplicate files across
|
// Command sfdupes quickly identifies candidate duplicate files across
|
||||||
// very large filesystems without reading full file contents. Files are
|
// very large filesystems without reading every byte of every file.
|
||||||
// considered duplicates when they have identical size, identical SHA-256
|
// Files are considered duplicates when their sizes are equal and they
|
||||||
// of their first 1024 bytes, and identical SHA-256 of their last 1024
|
// agree on a short ladder of SHA-256 hashes. A file under 10 MiB is
|
||||||
// bytes. scan maintains a persistent SQLite database of file signatures
|
// hashed in full. A larger file is compared on the hashes of its first
|
||||||
// (SFDUPES_DATABASE, default /var/lib/sfdupes/db.sqlite) that the
|
// and last 64 KiB, and only when those match another file's is its
|
||||||
// reporting subcommands read.
|
// content hash computed and compared: of the whole file when it is
|
||||||
|
// under 50 MiB, or of gigabyte-spaced 1 MiB samples when it is 50 MiB
|
||||||
|
// or larger. scan maintains a persistent SQLite database of file
|
||||||
|
// signatures (SFDUPES_DATABASE, default /var/lib/sfdupes/db.sqlite)
|
||||||
|
// that the reporting subcommands read.
|
||||||
//
|
//
|
||||||
// Usage:
|
// Usage:
|
||||||
//
|
//
|
||||||
@@ -97,7 +101,7 @@ func run(args []string, stderr io.Writer) int {
|
|||||||
func newRootCommand(stderr io.Writer) *cobra.Command {
|
func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||||
root := &cobra.Command{
|
root := &cobra.Command{
|
||||||
Use: "sfdupes",
|
Use: "sfdupes",
|
||||||
Short: "Find candidate duplicate files by size and head/tail SHA-256",
|
Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
|
||||||
Version: Version,
|
Version: Version,
|
||||||
Args: cobra.NoArgs,
|
Args: cobra.NoArgs,
|
||||||
RunE: func(cmd *cobra.Command, _ []string) error {
|
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||||
@@ -127,7 +131,7 @@ func newRootCommand(stderr io.Writer) *cobra.Command {
|
|||||||
}),
|
}),
|
||||||
}
|
}
|
||||||
scanCmd.Flags().IntVar(&scanWorkers, "workers", runtime.NumCPU(),
|
scanCmd.Flags().IntVar(&scanWorkers, "workers", runtime.NumCPU(),
|
||||||
"concurrent workers for the walk and hash phases")
|
"concurrent workers for the walk, hash, and content phases")
|
||||||
scanCmd.Flags().BoolVarP(&scanOneFS, "one-file-system", "x", false,
|
scanCmd.Flags().BoolVarP(&scanOneFS, "one-file-system", "x", false,
|
||||||
"do not cross filesystem boundaries")
|
"do not cross filesystem boundaries")
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +0,0 @@
|
|||||||
{
|
|
||||||
"devDependencies": {
|
|
||||||
"prettier": "3.8.1"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -17,13 +17,14 @@ const ioBufSize = 1 << 20
|
|||||||
const minGroupSize = 2
|
const minGroupSize = 2
|
||||||
|
|
||||||
// scanRec is one file record from the database. The signature (size,
|
// scanRec is one file record from the database. The signature (size,
|
||||||
// head, tail) is the duplicate key; mtime is informational only and
|
// head, tail, content) is the duplicate key; mtime is informational
|
||||||
// used by scan for change detection.
|
// only and used by scan for change detection.
|
||||||
type scanRec struct {
|
type scanRec struct {
|
||||||
size int64
|
size int64
|
||||||
mtime int64
|
mtime int64
|
||||||
head string
|
head string
|
||||||
tail string
|
tail string
|
||||||
|
content string
|
||||||
path string
|
path string
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -52,8 +53,9 @@ func loadRecords(ctx context.Context) ([]scanRec, error) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// dupeGroup is one set of candidate-duplicate files: identical size,
|
// dupeGroup is one set of candidate-duplicate files: identical size,
|
||||||
// head hash, and tail hash. paths is sorted lexicographically; the
|
// head hash, tail hash, and content hash. paths is sorted
|
||||||
// first entry is the group's "first", the rest are dupes.
|
// lexicographically; the first entry is the group's "first", the rest
|
||||||
|
// are dupes.
|
||||||
type dupeGroup struct {
|
type dupeGroup struct {
|
||||||
size int64
|
size int64
|
||||||
paths []string
|
paths []string
|
||||||
@@ -115,14 +117,15 @@ func collectDupeGroups(recs []scanRec) []dupeGroup {
|
|||||||
groups := make(map[fileSig][]string)
|
groups := make(map[fileSig][]string)
|
||||||
|
|
||||||
for _, r := range recs {
|
for _, r := range recs {
|
||||||
// A record without hashes (its size was unique when last
|
// A record without a content hash has unknown content and is
|
||||||
// scanned) has unknown content and is never reported as a
|
// never reported as a duplicate (README "Database").
|
||||||
// duplicate.
|
if r.content == "" {
|
||||||
if r.head == "" {
|
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|
||||||
k := fileSig{size: r.size, head: r.head, tail: r.tail}
|
k := fileSig{
|
||||||
|
size: r.size, head: r.head, tail: r.tail, content: r.content,
|
||||||
|
}
|
||||||
groups[k] = append(groups[k], r.path)
|
groups[k] = append(groups[k], r.path)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+43
-17
@@ -9,15 +9,15 @@ func TestCollectDupeGroups(t *testing.T) {
|
|||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 100, head: "h", tail: "t", path: "/z/b"},
|
{size: 100, head: "h", tail: "t", content: "c", path: "/z/b"},
|
||||||
{size: 100, head: "h", tail: "t", path: "/z/a"},
|
{size: 100, head: "h", tail: "t", content: "c", path: "/z/a"},
|
||||||
{size: 100, head: "h", tail: "t", path: "/z/c"},
|
{size: 100, head: "h", tail: "t", content: "c", path: "/z/c"},
|
||||||
{size: 4000, head: "H", tail: "T", path: "/big/2"},
|
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/2"},
|
||||||
{size: 4000, head: "H", tail: "T", path: "/big/1"},
|
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/1"},
|
||||||
// Same size as the /z group but a different head hash.
|
// Same size as the /z group but a different head hash.
|
||||||
{size: 100, head: "other", tail: "t", path: "/z/d"},
|
{size: 100, head: "other", tail: "t", content: "c", path: "/z/d"},
|
||||||
// A singleton signature must not form a group.
|
// A singleton signature must not form a group.
|
||||||
{size: 7, head: "u", tail: "u", path: "/lonely"},
|
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := collectDupeGroups(recs)
|
groups := collectDupeGroups(recs)
|
||||||
@@ -38,14 +38,40 @@ func TestCollectDupeGroups(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestCollectDupeGroupsContentSeparates(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
// Same size, head, and tail, but different content hashes: the final
|
||||||
|
// rung keeps them apart, so no group forms. Matching content groups.
|
||||||
|
// Records without a content hash never group, not even with each
|
||||||
|
// other.
|
||||||
|
recs := []scanRec{
|
||||||
|
{size: 100, head: "h", tail: "t", content: "c1", path: "/a"},
|
||||||
|
{size: 100, head: "h", tail: "t", content: "c2", path: "/b"},
|
||||||
|
{size: 100, head: "h", tail: "t", content: "c1", path: "/c"},
|
||||||
|
{size: 100, head: "h", tail: "t", path: "/d"},
|
||||||
|
{size: 100, head: "h", tail: "t", path: "/e"},
|
||||||
|
}
|
||||||
|
|
||||||
|
groups := collectDupeGroups(recs)
|
||||||
|
if len(groups) != 1 {
|
||||||
|
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
||||||
|
len(groups))
|
||||||
|
}
|
||||||
|
|
||||||
|
if !slices.Equal(groups[0].paths, []string{"/a", "/c"}) {
|
||||||
|
t.Errorf("group paths = %q, want /a /c", groups[0].paths)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
|
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// mtime is informational only; records differing only in mtime
|
// mtime is informational only; records differing only in mtime
|
||||||
// still group together.
|
// still group together.
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 9, mtime: 100, head: "h", tail: "t", path: "/m/1"},
|
{size: 9, mtime: 100, head: "h", tail: "t", content: "c", path: "/m/1"},
|
||||||
{size: 9, mtime: 200, head: "h", tail: "t", path: "/m/2"},
|
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := collectDupeGroups(recs)
|
groups := collectDupeGroups(recs)
|
||||||
@@ -58,10 +84,10 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
|||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 50, head: "b", tail: "b", path: "/beta/2"},
|
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"},
|
||||||
{size: 50, head: "b", tail: "b", path: "/beta/1"},
|
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"},
|
||||||
{size: 50, head: "a", tail: "a", path: "/alpha/2"},
|
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"},
|
||||||
{size: 50, head: "a", tail: "a", path: "/alpha/1"},
|
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := collectDupeGroups(recs)
|
groups := collectDupeGroups(recs)
|
||||||
@@ -80,10 +106,10 @@ func TestCollectDupeGroupsDeterministic(t *testing.T) {
|
|||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 1, head: "a", tail: "a", path: "/p/1"},
|
{size: 1, head: "a", tail: "a", content: "a", path: "/p/1"},
|
||||||
{size: 1, head: "a", tail: "a", path: "/p/2"},
|
{size: 1, head: "a", tail: "a", content: "a", path: "/p/2"},
|
||||||
{size: 2, head: "b", tail: "b", path: "/q/1"},
|
{size: 2, head: "b", tail: "b", content: "b", path: "/q/1"},
|
||||||
{size: 2, head: "b", tail: "b", path: "/q/2"},
|
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
||||||
}
|
}
|
||||||
|
|
||||||
forward := collectDupeGroups(recs)
|
forward := collectDupeGroups(recs)
|
||||||
|
|||||||
@@ -6,7 +6,9 @@ import (
|
|||||||
"crypto/sha256"
|
"crypto/sha256"
|
||||||
"database/sql"
|
"database/sql"
|
||||||
"encoding/hex"
|
"encoding/hex"
|
||||||
|
"errors"
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"io"
|
||||||
"io/fs"
|
"io/fs"
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
@@ -16,8 +18,40 @@ import (
|
|||||||
"syscall"
|
"syscall"
|
||||||
)
|
)
|
||||||
|
|
||||||
// chunk is the number of bytes hashed from each end of a file.
|
// The duplicate ladder (see hashSignature and README "Duplicate
|
||||||
const chunk = 1024
|
// detection"). A same-size candidate below headTailMin is hashed in
|
||||||
|
// full and compared directly; a larger one is separated first by the
|
||||||
|
// hashes of its end windows, then by a content hash that is exact below
|
||||||
|
// wholeFileMax and deliberately sampled at or above it. The hash phase
|
||||||
|
// reads only the end windows of a larger file; the content phase reads
|
||||||
|
// it for its content hash only once its size, head, and tail match
|
||||||
|
// another file's.
|
||||||
|
|
||||||
|
// headTailMin is the size threshold for the end-window gate. A file
|
||||||
|
// smaller than this is hashed in full directly, with no separate head
|
||||||
|
// and tail step: its head, tail, and content all carry the whole-file
|
||||||
|
// hash. A file this size or larger is separated first by its end
|
||||||
|
// windows.
|
||||||
|
const headTailMin = 10 * 1024 * 1024
|
||||||
|
|
||||||
|
// headTailWindow is the number of bytes hashed from each end of a file
|
||||||
|
// at or above headTailMin (the head and tail rungs). Because
|
||||||
|
// headTailMin is far larger than two windows, the head and tail windows
|
||||||
|
// never overlap.
|
||||||
|
const headTailWindow = 64 * 1024
|
||||||
|
|
||||||
|
// wholeFileMax is the size boundary between the two content rungs: a
|
||||||
|
// file strictly smaller than this is content-hashed in full; a file
|
||||||
|
// this size or larger is content-hashed by sampling.
|
||||||
|
const wholeFileMax = 50 * 1024 * 1024
|
||||||
|
|
||||||
|
// sampleStride is the spacing between content samples for large files:
|
||||||
|
// one window is read at each gigabyte-aligned offset (0, 1 GiB, ...).
|
||||||
|
const sampleStride = 1024 * 1024 * 1024
|
||||||
|
|
||||||
|
// sampleWindow is the number of bytes read at each large-file sample
|
||||||
|
// offset, truncated at end of file.
|
||||||
|
const sampleWindow = 1024 * 1024
|
||||||
|
|
||||||
// workQueueDepth bounds the job and result channels feeding the walk
|
// workQueueDepth bounds the job and result channels feeding the walk
|
||||||
// and hash worker pools.
|
// and hash worker pools.
|
||||||
@@ -44,16 +78,17 @@ type fileMeta struct {
|
|||||||
hashed bool
|
hashed bool
|
||||||
}
|
}
|
||||||
|
|
||||||
// runScan implements the scan subcommand: three sequential phases —
|
// runScan implements the scan subcommand: four sequential phases —
|
||||||
// walk (which stats each file as it is discovered), hash, update —
|
// walk (which stats each file as it is discovered), hash, update,
|
||||||
// that synchronize the persistent database with the filesystem state
|
// content — that synchronize the persistent database with the
|
||||||
// under the PATH operands. Only files whose size at least one other
|
// filesystem state under the PATH operands. Only files whose size at
|
||||||
// file shares are ever hashed: a size-unique file cannot be a
|
// least one other file shares are ever hashed: a size-unique file
|
||||||
// duplicate. Flag parsing and the at-least-one-operand check are done
|
// cannot be a duplicate. A file of headTailMin or more gets its content
|
||||||
// by cobra. Errors are returned rather than exiting, so that the
|
// hash only when its size, head, and tail match another file's. Flag
|
||||||
// deferred close — which checkpoints the SQLite WAL — always runs.
|
// parsing and the at-least-one-operand check are done by cobra. Errors
|
||||||
// Cancelling ctx unwinds the worker pools and aborts the scan with the
|
// are returned rather than exiting, so that the deferred close — which
|
||||||
// context's error.
|
// checkpoints the SQLite WAL — always runs. Cancelling ctx unwinds the
|
||||||
|
// worker pools and aborts the scan with the context's error.
|
||||||
func runScan(ctx context.Context, roots []string, workers int,
|
func runScan(ctx context.Context, roots []string, workers int,
|
||||||
oneFS bool,
|
oneFS bool,
|
||||||
) error {
|
) error {
|
||||||
@@ -166,13 +201,16 @@ type scanState struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// syncScan synchronizes the database with the filesystem under roots
|
// syncScan synchronizes the database with the filesystem under roots
|
||||||
// in three sequential phases: walk (enumerate and stat every file,
|
// in four sequential phases: walk (enumerate and stat every file,
|
||||||
// building a complete size census), hash (read only the new or
|
// building a complete size census), hash (read only the new or
|
||||||
// changed — or previously unhashed — files whose size at least one
|
// changed — or previously unhashed — files whose size at least one
|
||||||
// other file shares, committing results in batches as they arrive),
|
// other file shares, committing results in batches as they arrive),
|
||||||
// and update (record the size-unique files without reading them, and
|
// update (record the size-unique files without reading them, and
|
||||||
// delete the records the scan no longer verifies). Records outside
|
// delete the records the scan no longer verifies), and content (fill
|
||||||
// the roots are never touched.
|
// in the content hash of every record of headTailMin or more whose
|
||||||
|
// size, head, and tail match another record's). Records outside the
|
||||||
|
// roots are never touched, except that the content phase fills in
|
||||||
|
// their content hash.
|
||||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||||
workers int, oneFS bool,
|
workers int, oneFS bool,
|
||||||
) (scanStats, error) {
|
) (scanStats, error) {
|
||||||
@@ -207,7 +245,12 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
|||||||
return s.st, err
|
return s.st, err
|
||||||
}
|
}
|
||||||
|
|
||||||
return s.st, s.updatePhase(ctx)
|
err = s.updatePhase(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return s.st, err
|
||||||
|
}
|
||||||
|
|
||||||
|
return s.st, s.contentPhase(ctx, workers)
|
||||||
}
|
}
|
||||||
|
|
||||||
// loadIndex indexes the database records under the scan roots for
|
// loadIndex indexes the database records under the scan roots for
|
||||||
@@ -241,7 +284,8 @@ func (s *scanState) loadIndex(ctx context.Context, roots []string) error {
|
|||||||
|
|
||||||
// walkPhase drains the walk, appending every walked file's size to
|
// walkPhase drains the walk, appending every walked file's size to
|
||||||
// the census and resolving what it can immediately: an unchanged file
|
// the census and resolving what it can immediately: an unchanged file
|
||||||
// whose record already has hashes needs nothing further. It returns
|
// whose record already has hashes needs nothing from the hash phase
|
||||||
|
// (the content phase may still fill in its content hash). It returns
|
||||||
// the new-or-changed files and the unchanged files whose records lack
|
// the new-or-changed files and the unchanged files whose records lack
|
||||||
// hashes; both remain candidates until the census decides whether
|
// hashes; both remain candidates until the census decides whether
|
||||||
// their sizes are shared.
|
// their sizes are shared.
|
||||||
@@ -380,27 +424,40 @@ func sameInode(a, b fileRec) bool {
|
|||||||
return (a.dev != 0 || a.ino != 0) && a.dev == b.dev && a.ino == b.ino
|
return (a.dev != 0 || a.ino != 0) && a.dev == b.dev && a.ino == b.ino
|
||||||
}
|
}
|
||||||
|
|
||||||
// hashPhase hashes every queued file with the worker pool — one read
|
// hashPhase hashes every queued file with hashSignature — the head and
|
||||||
// per inode run, in inode order — committing completed records to the
|
// tail of a file of headTailMin or more, the whole file below that —
|
||||||
// database in batches as results arrive, so a long scan persists its
|
// committing completed records to the database in batches as results
|
||||||
// progress as it goes (an interrupted scan resumes cheaply: the next
|
// arrive, so a long scan persists its progress as it goes (an
|
||||||
// run skips everything already recorded). The total counts actual
|
// interrupted scan resumes cheaply: the next run skips everything
|
||||||
// reads, so the bar shows a real ETA. A run that fails to hash is
|
// already recorded). A run that fails to hash is warned about and
|
||||||
// warned about and skipped; stale records for its paths, if any, are
|
// skipped; stale records for its paths, if any, are deleted by the
|
||||||
// deleted by the update phase.
|
// update phase.
|
||||||
|
func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
||||||
|
runs := hashRuns(s.toHash)
|
||||||
|
s.toHash = nil
|
||||||
|
|
||||||
|
return s.readRuns(ctx, workers, "hash", runs, hashSignature, s.recordRun)
|
||||||
|
}
|
||||||
|
|
||||||
|
// readRuns reads runs with the worker pool, one read per inode run, in
|
||||||
|
// the order given, under a progress display named label. The workers
|
||||||
|
// compute each run's hashes with hash, and each result goes to record;
|
||||||
|
// a run that fails to read is warned about and counted as skipped
|
||||||
|
// instead. The total counts actual reads, so the bar shows a real ETA.
|
||||||
//
|
//
|
||||||
// Returning early — a failed database write, or a cancelled scan — must
|
// Returning early — a failed database write, or a cancelled scan — must
|
||||||
// not strand the pool: the feeder would park forever on a full jobs
|
// not strand the pool: the feeder would park forever on a full jobs
|
||||||
// channel and every worker on a full results channel. The deferred stop
|
// channel and every worker on a full results channel. The deferred stop
|
||||||
// is what prevents that.
|
// is what prevents that.
|
||||||
func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
func (s *scanState) readRuns(ctx context.Context, workers int,
|
||||||
runs := hashRuns(s.toHash)
|
label string, runs [][]fileRec,
|
||||||
s.toHash = nil
|
hash func(path string, size int64) (string, string, string, error),
|
||||||
|
record func(ctx context.Context, r hashResult) error,
|
||||||
pool := startHashPool(ctx, runs, workers)
|
) error {
|
||||||
|
pool := startHashPool(ctx, runs, workers, hash)
|
||||||
defer pool.stop()
|
defer pool.stop()
|
||||||
|
|
||||||
prog := newProgress("hash", int64(len(runs)))
|
prog := newProgress(label, int64(len(runs)))
|
||||||
defer prog.finish()
|
defer prog.finish()
|
||||||
|
|
||||||
for range runs {
|
for range runs {
|
||||||
@@ -417,12 +474,12 @@ func (s *scanState) hashPhase(ctx context.Context, workers int) error {
|
|||||||
if r.err != nil {
|
if r.err != nil {
|
||||||
s.st.skipped += len(r.run)
|
s.st.skipped += len(r.run)
|
||||||
|
|
||||||
prog.warnf("hash %s: %v", r.run[0].path, r.err)
|
prog.warnf("%s %s: %v", label, r.run[0].path, r.err)
|
||||||
|
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|
||||||
err := s.recordRun(ctx, r)
|
err := record(ctx, r)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
@@ -443,10 +500,17 @@ func (s *scanState) recordRun(ctx context.Context, r hashResult) error {
|
|||||||
mtime: rec.mtime,
|
mtime: rec.mtime,
|
||||||
head: r.head,
|
head: r.head,
|
||||||
tail: r.tail,
|
tail: r.tail,
|
||||||
|
content: r.content,
|
||||||
path: rec.path,
|
path: rec.path,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
return s.commitFullBatch(ctx)
|
||||||
|
}
|
||||||
|
|
||||||
|
// commitFullBatch commits the running batch once it holds
|
||||||
|
// updateBatchSize records.
|
||||||
|
func (s *scanState) commitFullBatch(ctx context.Context) error {
|
||||||
if len(s.batch) < updateBatchSize {
|
if len(s.batch) < updateBatchSize {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
@@ -504,6 +568,147 @@ func (s *scanState) updatePhase(ctx context.Context) error {
|
|||||||
return applyChanges(ctx, s.db, nil, deletes, prog)
|
return applyChanges(ctx, s.db, nil, deletes, prog)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// contentPhase fills in the content hash of every record of headTailMin
|
||||||
|
// or more that lacks one and whose size, head, and tail equal another
|
||||||
|
// record's, anywhere in the database: records from this scan and
|
||||||
|
// records stored by earlier scans, inside or outside the roots. Only
|
||||||
|
// such a file can still be a duplicate, so no other file of headTailMin
|
||||||
|
// or more is read beyond its end windows. The files are read with the
|
||||||
|
// hash phase's worker pool and their records written back in batches. A
|
||||||
|
// failed read is warned about and counted as skipped; the record keeps
|
||||||
|
// its empty content, so it is never grouped, and a later scan tries
|
||||||
|
// again.
|
||||||
|
func (s *scanState) contentPhase(ctx context.Context, workers int) error {
|
||||||
|
toRead, recs, err := s.contentCandidates(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
err = s.readRuns(ctx, workers, "content", hashRuns(toRead),
|
||||||
|
hashContentOnly, func(ctx context.Context, r hashResult) error {
|
||||||
|
// Every path in the run keeps its record's head and tail
|
||||||
|
// and gains the one content hash read for the run.
|
||||||
|
for _, f := range r.run {
|
||||||
|
rec := recs[f.path]
|
||||||
|
rec.content = r.content
|
||||||
|
s.batch = append(s.batch, rec)
|
||||||
|
}
|
||||||
|
|
||||||
|
return s.commitFullBatch(ctx)
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
return applyChanges(ctx, s.db, s.batch, nil, nil)
|
||||||
|
}
|
||||||
|
|
||||||
|
// contentCandidates returns the files the content phase reads, and
|
||||||
|
// their records by path. Every record contentCandidatesSQL returns has
|
||||||
|
// its file checked with lstat, whether or not it already has a content
|
||||||
|
// hash: a file that is gone, is no longer a regular file, or has
|
||||||
|
// changed by the walk's rule keeps its record as it is and does not
|
||||||
|
// count as a match for the others, and any other lstat error is warned
|
||||||
|
// about and counted as skipped, with the same result. If such a record
|
||||||
|
// has no content hash, it stays out of duplicate groups; if it has one,
|
||||||
|
// it is still reported until a scan covering its own tree updates or
|
||||||
|
// removes it. The files of a group that pass and have no content hash
|
||||||
|
// are read only if at least minGroupSize of the group's files pass, so
|
||||||
|
// a group whose other members are all stale costs no reads. Only the
|
||||||
|
// records to be read are kept.
|
||||||
|
func (s *scanState) contentCandidates(
|
||||||
|
ctx context.Context,
|
||||||
|
) ([]fileRec, map[string]scanRec, error) {
|
||||||
|
// The query and the checks take real time on a large database;
|
||||||
|
// without a display the scan looks hung before the reads begin.
|
||||||
|
prog := newProgress("content", -1)
|
||||||
|
defer prog.finish()
|
||||||
|
|
||||||
|
var (
|
||||||
|
toRead []fileRec
|
||||||
|
first scanRec // the current group's first record
|
||||||
|
passed int // the current group's files that passed the check
|
||||||
|
unread []fileRec // those of them without a content hash
|
||||||
|
)
|
||||||
|
|
||||||
|
recs := make(map[string]scanRec)
|
||||||
|
|
||||||
|
// endGroup queues the current group's files to read if at least
|
||||||
|
// minGroupSize of its files passed, and drops their records if not.
|
||||||
|
endGroup := func() {
|
||||||
|
if passed >= minGroupSize {
|
||||||
|
toRead = append(toRead, unread...)
|
||||||
|
} else {
|
||||||
|
for _, f := range unread {
|
||||||
|
delete(recs, f.path)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
passed, unread = 0, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
err := loadContentCandidates(ctx, s.db, func(r scanRec, hashed bool) {
|
||||||
|
prog.increment()
|
||||||
|
|
||||||
|
if r.size != first.size || r.head != first.head || r.tail != first.tail {
|
||||||
|
endGroup()
|
||||||
|
|
||||||
|
first = r
|
||||||
|
}
|
||||||
|
|
||||||
|
f, ok, err := unchangedFile(r)
|
||||||
|
if err != nil {
|
||||||
|
s.st.skipped++
|
||||||
|
|
||||||
|
prog.warnf("content %s: %v", r.path, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
passed++
|
||||||
|
|
||||||
|
if !hashed {
|
||||||
|
unread = append(unread, f)
|
||||||
|
recs[r.path] = r
|
||||||
|
}
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
endGroup()
|
||||||
|
|
||||||
|
return toRead, recs, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// unchangedFile lstats the file r names and returns it for reading if
|
||||||
|
// it is still the regular file r records: the same size, and an mtime
|
||||||
|
// no newer than recorded (the walk's change rule). A file that is gone
|
||||||
|
// or has changed reports false; any other lstat error is returned.
|
||||||
|
func unchangedFile(r scanRec) (fileRec, bool, error) {
|
||||||
|
fi, err := os.Lstat(r.path)
|
||||||
|
if errors.Is(err, fs.ErrNotExist) {
|
||||||
|
return fileRec{}, false, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
return fileRec{}, false, err
|
||||||
|
}
|
||||||
|
|
||||||
|
if !fi.Mode().IsRegular() || fi.Size() != r.size ||
|
||||||
|
fi.ModTime().Unix() > r.mtime {
|
||||||
|
return fileRec{}, false, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
dev, ino := inodeOfInfo(fi)
|
||||||
|
|
||||||
|
return fileRec{
|
||||||
|
path: r.path, size: r.size, mtime: r.mtime, dev: dev, ino: ino,
|
||||||
|
}, true, nil
|
||||||
|
}
|
||||||
|
|
||||||
// underAnyRoot reports whether path is any of the roots or lies under
|
// underAnyRoot reports whether path is any of the roots or lies under
|
||||||
// one of them.
|
// one of them.
|
||||||
func underAnyRoot(path string, roots []string) bool {
|
func underAnyRoot(path string, roots []string) bool {
|
||||||
@@ -834,12 +1039,15 @@ func inodeOfInfo(fi fs.FileInfo) (uint64, uint64) {
|
|||||||
return statDev(st), st.Ino
|
return statDev(st), st.Ino
|
||||||
}
|
}
|
||||||
|
|
||||||
// hashResult carries one inode run's head/tail hashes (or the error
|
// hashResult carries the hashes computed for one inode run (or the
|
||||||
// that prevented hashing it) from the hash workers to the hash phase.
|
// error that prevented computing them) from the pool's workers to the
|
||||||
|
// phase that started the pool: head, tail, and content from
|
||||||
|
// hashSignature, content alone from hashContentOnly.
|
||||||
type hashResult struct {
|
type hashResult struct {
|
||||||
run []fileRec
|
run []fileRec
|
||||||
head string
|
head string
|
||||||
tail string
|
tail string
|
||||||
|
content string
|
||||||
err error
|
err error
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -856,10 +1064,10 @@ type hashPool struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// startHashPool starts the feeder and the workers over runs. Workers
|
// startHashPool starts the feeder and the workers over runs. Workers
|
||||||
// hash each run's first path (all paths in a run are hard links to the
|
// hash each run's first path with hash (all paths in a run are hard
|
||||||
// same inode) and write one result per run.
|
// links to the same inode) and write one result per run.
|
||||||
func startHashPool(ctx context.Context, runs [][]fileRec,
|
func startHashPool(ctx context.Context, runs [][]fileRec, workers int,
|
||||||
workers int,
|
hash func(path string, size int64) (string, string, string, error),
|
||||||
) *hashPool {
|
) *hashPool {
|
||||||
ctx, cancel := context.WithCancel(ctx)
|
ctx, cancel := context.WithCancel(ctx)
|
||||||
|
|
||||||
@@ -871,7 +1079,7 @@ func startHashPool(ctx context.Context, runs [][]fileRec,
|
|||||||
wg.Go(func() { feedHashJobs(ctx, runs, jobs) })
|
wg.Go(func() { feedHashJobs(ctx, runs, jobs) })
|
||||||
|
|
||||||
for range workers {
|
for range workers {
|
||||||
wg.Go(func() { hashWorker(ctx, jobs, results) })
|
wg.Go(func() { hashWorker(ctx, jobs, results, hash) })
|
||||||
}
|
}
|
||||||
|
|
||||||
done := make(chan struct{})
|
done := make(chan struct{})
|
||||||
@@ -917,24 +1125,25 @@ func feedHashJobs(ctx context.Context, runs [][]fileRec,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// hashWorker hashes one inode run at a time until jobs is closed or the
|
// hashWorker hashes one inode run at a time with hash until jobs is
|
||||||
// scan is cancelled. A cancelled worker drops the runs still queued
|
// closed or the scan is cancelled. A cancelled worker drops the runs
|
||||||
// instead of stopping its reads of jobs: the range must run out for the
|
// still queued instead of stopping its reads of jobs: the range must
|
||||||
// pool to tear down, and reading a file nobody wants the hash of only
|
// run out for the pool to tear down, and reading a file nobody wants
|
||||||
// delays that.
|
// the hash of only delays that.
|
||||||
func hashWorker(ctx context.Context, jobs <-chan []fileRec,
|
func hashWorker(ctx context.Context, jobs <-chan []fileRec,
|
||||||
results chan<- hashResult,
|
results chan<- hashResult,
|
||||||
|
hash func(path string, size int64) (string, string, string, error),
|
||||||
) {
|
) {
|
||||||
for run := range jobs {
|
for run := range jobs {
|
||||||
if ctx.Err() != nil {
|
if ctx.Err() != nil {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|
||||||
head, tail, err := hashHeadTail(run[0].path, run[0].size)
|
head, tail, content, err := hash(run[0].path, run[0].size)
|
||||||
|
|
||||||
select {
|
select {
|
||||||
case results <- hashResult{
|
case results <- hashResult{
|
||||||
run: run, head: head, tail: tail, err: err,
|
run: run, head: head, tail: tail, content: content, err: err,
|
||||||
}:
|
}:
|
||||||
case <-ctx.Done():
|
case <-ctx.Done():
|
||||||
return
|
return
|
||||||
@@ -942,55 +1151,151 @@ func hashWorker(ctx context.Context, jobs <-chan []fileRec,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// emptyHash is the lowercase-hex SHA-256 of the empty input: the head
|
// emptyHash is the lowercase-hex SHA-256 of the empty input: the head,
|
||||||
// and tail hash of every zero-length file.
|
// tail, and content hash of every zero-length file.
|
||||||
const emptyHash = "e3b0c44298fc1c149afbf4c8996fb924" +
|
const emptyHash = "e3b0c44298fc1c149afbf4c8996fb924" +
|
||||||
"27ae41e4649b934ca495991b7852b855"
|
"27ae41e4649b934ca495991b7852b855"
|
||||||
|
|
||||||
// hashHeadTail returns the lowercase-hex SHA-256 of the first
|
// hashSignature computes the hashes the hash phase records for a file
|
||||||
// min(chunk, size) bytes and of the last min(chunk, size) bytes of the
|
// whose size is shared; with the file size they form its duplicate
|
||||||
// file at path. The two reads overlap when size < 2*chunk. size is the
|
// signature. A file below headTailMin is hashed in full and its
|
||||||
// value recorded when the file was statted; a zero-length file's
|
// whole-file SHA-256 is returned as head, tail, and content alike —
|
||||||
// hashes are constant, so it is never even opened.
|
// that range takes no separate end-window step. For a file at or above
|
||||||
func hashHeadTail(path string, size int64) (string, string, error) {
|
// headTailMin only the head and tail are computed, the SHA-256 of its
|
||||||
|
// first and last headTailWindow bytes, and content is returned empty:
|
||||||
|
// the content phase computes it with hashContentOnly once the file's
|
||||||
|
// size, head, and tail match another file's. Two files are duplicates
|
||||||
|
// only when all four agree; any mismatch means not a duplicate. size
|
||||||
|
// is the value recorded when the file was statted; a zero-length file
|
||||||
|
// has constant hashes and is never opened.
|
||||||
|
func hashSignature(path string, size int64) (string, string, string, error) {
|
||||||
if size == 0 {
|
if size == 0 {
|
||||||
return emptyHash, emptyHash, nil
|
return emptyHash, emptyHash, emptyHash, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
//nolint:gosec // hashing operator-supplied paths is the tool's purpose
|
//nolint:gosec // hashing operator-supplied paths is the tool's purpose
|
||||||
f, err := os.Open(path)
|
f, err := os.Open(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return "", "", err
|
return "", "", "", err
|
||||||
}
|
}
|
||||||
|
|
||||||
defer func() { _ = f.Close() }()
|
defer func() { _ = f.Close() }()
|
||||||
|
|
||||||
n := min(int64(chunk), size)
|
// Below the threshold the whole file is hashed directly, with no
|
||||||
|
// end-window step: head and tail both carry the whole-file hash.
|
||||||
|
if size < int64(headTailMin) {
|
||||||
|
content, err := hashWhole(f, size)
|
||||||
|
if err != nil {
|
||||||
|
return "", "", "", err
|
||||||
|
}
|
||||||
|
|
||||||
buf := make([]byte, n)
|
return content, content, content, nil
|
||||||
|
}
|
||||||
|
|
||||||
_, err = f.ReadAt(buf, 0)
|
head, tail, err := hashEnds(f, size)
|
||||||
|
if err != nil {
|
||||||
|
return "", "", "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
return head, tail, "", nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// hashContentOnly returns the content hash of the file at path, which
|
||||||
|
// is at least headTailMin bytes: the content phase's read. head and
|
||||||
|
// tail are returned empty, because the content phase keeps the ones its
|
||||||
|
// records already hold.
|
||||||
|
func hashContentOnly(path string, size int64) (string, string, string, error) {
|
||||||
|
//nolint:gosec // hashing operator-supplied paths is the tool's purpose
|
||||||
|
f, err := os.Open(path)
|
||||||
|
if err != nil {
|
||||||
|
return "", "", "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
defer func() { _ = f.Close() }()
|
||||||
|
|
||||||
|
content, err := hashContent(f, size)
|
||||||
|
|
||||||
|
return "", "", content, err
|
||||||
|
}
|
||||||
|
|
||||||
|
// hashEnds returns the SHA-256 of the first and last headTailWindow
|
||||||
|
// bytes of f. It is called only for files at least headTailMin, which
|
||||||
|
// is far larger than two windows, so the windows never overlap and both
|
||||||
|
// reads are always full.
|
||||||
|
func hashEnds(f *os.File, size int64) (string, string, error) {
|
||||||
|
buf := make([]byte, headTailWindow)
|
||||||
|
|
||||||
|
_, err := f.ReadAt(buf, 0)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return "", "", err
|
return "", "", err
|
||||||
}
|
}
|
||||||
|
|
||||||
h := sha256.Sum256(buf)
|
h := sha256.Sum256(buf)
|
||||||
|
head := hex.EncodeToString(h[:])
|
||||||
|
|
||||||
// When the whole file fits in one chunk the tail window is exactly
|
_, err = f.ReadAt(buf, size-int64(headTailWindow))
|
||||||
// the bytes just read: reuse the head hash instead of issuing a
|
|
||||||
// second read for every small file.
|
|
||||||
if size <= int64(chunk) {
|
|
||||||
hh := hex.EncodeToString(h[:])
|
|
||||||
|
|
||||||
return hh, hh, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err = f.ReadAt(buf, size-n)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return "", "", err
|
return "", "", err
|
||||||
}
|
}
|
||||||
|
|
||||||
t := sha256.Sum256(buf)
|
t := sha256.Sum256(buf)
|
||||||
|
|
||||||
return hex.EncodeToString(h[:]), hex.EncodeToString(t[:]), nil
|
return head, hex.EncodeToString(t[:]), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// hashContent returns the content-rung hash of f: the SHA-256 of the
|
||||||
|
// whole file when it is smaller than wholeFileMax, or of sampled
|
||||||
|
// windows when it is that size or larger.
|
||||||
|
func hashContent(f *os.File, size int64) (string, error) {
|
||||||
|
if size >= int64(wholeFileMax) {
|
||||||
|
return hashSamples(f, size)
|
||||||
|
}
|
||||||
|
|
||||||
|
return hashWhole(f, size)
|
||||||
|
}
|
||||||
|
|
||||||
|
// hashWhole returns the SHA-256 of the entire file. A SectionReader is
|
||||||
|
// used so the read is independent of the offset left by any end-window
|
||||||
|
// reads. Reading fewer than size bytes means the file shrank between
|
||||||
|
// the stat and the hash; that is an error rather than a hash of content
|
||||||
|
// that no longer matches the recorded size.
|
||||||
|
func hashWhole(f *os.File, size int64) (string, error) {
|
||||||
|
h := sha256.New()
|
||||||
|
|
||||||
|
n, err := io.Copy(h, io.NewSectionReader(f, 0, size))
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
if n != size {
|
||||||
|
return "", fmt.Errorf("read %d of %d bytes: %w", n, size,
|
||||||
|
io.ErrUnexpectedEOF)
|
||||||
|
}
|
||||||
|
|
||||||
|
return hex.EncodeToString(h.Sum(nil)), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// hashSamples feeds sampleWindow bytes at each gigabyte-aligned offset
|
||||||
|
// (0, sampleStride, 2*sampleStride, ... while inside the file), in
|
||||||
|
// order, into one hash, each window truncated at end of file. This is
|
||||||
|
// the probabilistic large-file rung: two files of equal size agreeing
|
||||||
|
// on every sample are reported as duplicates without every byte being
|
||||||
|
// read. Because size is part of the signature, files of different sizes
|
||||||
|
// never reach this comparison, so the sample boundaries always align.
|
||||||
|
func hashSamples(f *os.File, size int64) (string, error) {
|
||||||
|
h := sha256.New()
|
||||||
|
buf := make([]byte, sampleWindow)
|
||||||
|
|
||||||
|
for off := int64(0); off < size; off += int64(sampleStride) {
|
||||||
|
n := min(int64(sampleWindow), size-off)
|
||||||
|
|
||||||
|
_, err := f.ReadAt(buf[:n], off)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
h.Write(buf[:n])
|
||||||
|
}
|
||||||
|
|
||||||
|
return hex.EncodeToString(h.Sum(nil)), nil
|
||||||
}
|
}
|
||||||
|
|||||||
+636
-40
@@ -53,7 +53,23 @@ func pattern(tag byte, n int) []byte {
|
|||||||
return data
|
return data
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestHashHeadTail(t *testing.T) {
|
// sig returns a file's full signature (head, tail, content), failing the
|
||||||
|
// test on any error.
|
||||||
|
func sig(t *testing.T, path string, size int64) (string, string, string) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
head, tail, content, err := hashSignature(path, size)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("hashSignature %s: %v", path, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return head, tail, content
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestHashSignatureBelowThreshold verifies that a file below headTailMin
|
||||||
|
// is hashed in full and compared directly: head, tail, and content all
|
||||||
|
// carry the whole-file SHA-256, with no separate end-window step.
|
||||||
|
func TestHashSignatureBelowThreshold(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
@@ -62,13 +78,10 @@ func TestHashHeadTail(t *testing.T) {
|
|||||||
name string
|
name string
|
||||||
data []byte
|
data []byte
|
||||||
}{
|
}{
|
||||||
{"empty", nil},
|
|
||||||
{"one-byte", []byte("x")},
|
{"one-byte", []byte("x")},
|
||||||
{"under-one-chunk", pattern(1, chunk-1)},
|
{"one-window", pattern(1, headTailWindow)},
|
||||||
{"exactly-one-chunk", pattern(2, chunk)},
|
{"several-windows", pattern(2, 3*headTailWindow)},
|
||||||
{"overlapping-reads", pattern(3, chunk+chunk/2)},
|
{"near-threshold", pattern(3, headTailMin-1)},
|
||||||
{"exactly-two-chunks", pattern(4, 2*chunk)},
|
|
||||||
{"beyond-two-chunks", pattern(5, 3*chunk)},
|
|
||||||
}
|
}
|
||||||
for _, c := range cases {
|
for _, c := range cases {
|
||||||
t.Run(c.name, func(t *testing.T) {
|
t.Run(c.name, func(t *testing.T) {
|
||||||
@@ -76,49 +89,619 @@ func TestHashHeadTail(t *testing.T) {
|
|||||||
|
|
||||||
p := writeFile(t, dir, c.name, c.data)
|
p := writeFile(t, dir, c.name, c.data)
|
||||||
|
|
||||||
head, tail, err := hashHeadTail(p, int64(len(c.data)))
|
head, tail, content := sig(t, p, int64(len(c.data)))
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("hashHeadTail: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
n := min(chunk, len(c.data))
|
whole := hexSum(c.data)
|
||||||
if want := hexSum(c.data[:n]); head != want {
|
if head != whole || tail != whole || content != whole {
|
||||||
t.Errorf("head = %s, want %s", head, want)
|
t.Errorf("head=%s tail=%s content=%s, want all whole-file %s",
|
||||||
}
|
head, tail, content, whole)
|
||||||
|
|
||||||
if want := hexSum(c.data[len(c.data)-n:]); tail != want {
|
|
||||||
t.Errorf("tail = %s, want %s", tail, want)
|
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestHashHeadTailErrors(t *testing.T) {
|
// TestHashSignatureEnds exercises the head and tail rungs, which apply
|
||||||
|
// only to files at least headTailMin. Sparse files keep the fixtures
|
||||||
|
// cheap: a difference in the first window changes only head, a
|
||||||
|
// difference in the last window changes only tail, and a difference
|
||||||
|
// between the windows changes neither end hash but does change the
|
||||||
|
// whole-file content rung (the file is below wholeFileMax).
|
||||||
|
// hashSignature leaves the content hash of a file this size to the
|
||||||
|
// content phase, so that rung is checked through a scan.
|
||||||
|
func TestHashSignatureEnds(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
dir := t.TempDir()
|
dir := t.TempDir()
|
||||||
|
|
||||||
_, _, err := hashHeadTail(filepath.Join(dir, "missing"), 1)
|
// Between headTailMin and wholeFileMax: the end-window gate is active
|
||||||
|
// and the content rung is a whole-file hash.
|
||||||
|
const size = int64(headTailMin + 2*1024*1024)
|
||||||
|
|
||||||
|
base := sparseFile(t, dir, "ends-base", size)
|
||||||
|
headDiff := sparseFile(t, dir, "ends-head", size)
|
||||||
|
tailDiff := sparseFile(t, dir, "ends-tail", size)
|
||||||
|
midDiff := sparseFile(t, dir, "ends-mid", size)
|
||||||
|
|
||||||
|
pokeAt(t, headDiff, 0, []byte{1})
|
||||||
|
pokeAt(t, tailDiff, size-1, []byte{1})
|
||||||
|
pokeAt(t, midDiff, size/2, []byte{1})
|
||||||
|
|
||||||
|
bHead, bTail, bContent := sig(t, base, size)
|
||||||
|
if bContent != "" {
|
||||||
|
t.Errorf("content = %q, want none from the hash phase", bContent)
|
||||||
|
}
|
||||||
|
|
||||||
|
h, tl, _ := sig(t, headDiff, size)
|
||||||
|
if h == bHead {
|
||||||
|
t.Error("a byte in the first window did not change head")
|
||||||
|
}
|
||||||
|
|
||||||
|
if tl != bTail {
|
||||||
|
t.Error("a byte in the first window changed tail")
|
||||||
|
}
|
||||||
|
|
||||||
|
h, tl, _ = sig(t, tailDiff, size)
|
||||||
|
if tl == bTail {
|
||||||
|
t.Error("a byte in the last window did not change tail")
|
||||||
|
}
|
||||||
|
|
||||||
|
if h != bHead {
|
||||||
|
t.Error("a byte in the last window changed head")
|
||||||
|
}
|
||||||
|
|
||||||
|
h, tl, _ = sig(t, midDiff, size)
|
||||||
|
if h != bHead || tl != bTail {
|
||||||
|
t.Error("a byte between the windows changed an end hash")
|
||||||
|
}
|
||||||
|
|
||||||
|
// base and midDiff match on size, head, and tail, so the scan reads
|
||||||
|
// both for their content hashes.
|
||||||
|
c := scanContents(t, dir, base, midDiff)
|
||||||
|
if c[midDiff] == c[base] {
|
||||||
|
t.Error("whole-file content rung ignored a byte between the windows")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestHashSignatureErrors(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
|
||||||
|
// A missing file: an error, and every hash left empty.
|
||||||
|
head, tail, content, err := hashSignature(filepath.Join(dir, "missing"), 1)
|
||||||
if err == nil {
|
if err == nil {
|
||||||
t.Error("no error for a missing file")
|
t.Error("no error for a missing file")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if head != "" || tail != "" || content != "" {
|
||||||
|
t.Errorf("missing file returned hashes: %q %q %q", head, tail, content)
|
||||||
|
}
|
||||||
|
|
||||||
// A zero-length file has constant hashes and is never opened: even
|
// A zero-length file has constant hashes and is never opened: even
|
||||||
// a missing path succeeds.
|
// a missing path succeeds.
|
||||||
head, tail, err := hashHeadTail(filepath.Join(dir, "missing"), 0)
|
head, tail, content, err = hashSignature(filepath.Join(dir, "missing"), 0)
|
||||||
if err != nil || head != emptyHash || tail != emptyHash {
|
if err != nil ||
|
||||||
t.Errorf("empty: head=%q tail=%q err=%v, want constant hashes",
|
head != emptyHash || tail != emptyHash || content != emptyHash {
|
||||||
head, tail, err)
|
t.Errorf("empty: head=%q tail=%q content=%q err=%v, "+
|
||||||
|
"want constant hashes", head, tail, content, err)
|
||||||
}
|
}
|
||||||
|
|
||||||
// A file that shrank between the stat and hash passes: reading at
|
// A file that shrank between the stat and hash passes: reading at
|
||||||
// the stat-reported size must fail rather than emit wrong hashes.
|
// the stat-reported size must fail rather than emit wrong hashes.
|
||||||
p := writeFile(t, dir, "shrunk", []byte("tiny"))
|
p := writeFile(t, dir, "shrunk", []byte("tiny"))
|
||||||
|
|
||||||
_, _, err = hashHeadTail(p, int64(2*chunk))
|
head, tail, content, err = hashSignature(p, int64(2*headTailWindow))
|
||||||
if err == nil {
|
if err == nil {
|
||||||
t.Error("no error when the stat size exceeds the file size")
|
t.Error("no error when the stat size exceeds the file size")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if head != "" || tail != "" || content != "" {
|
||||||
|
t.Errorf("shrunk file returned hashes: %q %q %q", head, tail, content)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// sparseFile creates a file that is logically size bytes long without
|
||||||
|
// allocating blocks for the hole, so multi-gigabyte cases stay cheap.
|
||||||
|
func sparseFile(t *testing.T, dir, name string, size int64) string {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
p := filepath.Join(dir, name)
|
||||||
|
|
||||||
|
f, err := os.Create(p) //nolint:gosec // test-controlled path
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = f.Truncate(size)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = f.Close()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return p
|
||||||
|
}
|
||||||
|
|
||||||
|
// pokeAt writes data into an existing file at off, leaving the rest of
|
||||||
|
// the file (a sparse hole) untouched.
|
||||||
|
func pokeAt(t *testing.T, path string, off int64, data []byte) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
f, err := os.OpenFile(path, os.O_WRONLY, 0o600) //nolint:gosec // test path
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
_, err = f.WriteAt(data, off)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = f.Close()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// scanContents scans dir into a fresh database and returns the content
|
||||||
|
// hash recorded for each file, by path, failing the test if one of want
|
||||||
|
// has none. A file of headTailMin or more gets a content hash only when
|
||||||
|
// it is scanned with a file of the same size, head, and tail.
|
||||||
|
func scanContents(t *testing.T, dir string,
|
||||||
|
want ...string,
|
||||||
|
) map[string]string {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
syncTree(t, db, dir)
|
||||||
|
|
||||||
|
contents := make(map[string]string)
|
||||||
|
for _, r := range dbRecords(t, db) {
|
||||||
|
contents[r.path] = r.content
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, p := range want {
|
||||||
|
if contents[p] == "" {
|
||||||
|
t.Fatalf("%s: no content hash", p)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return contents
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestContentRungBoundary checks the 50 MiB boundary between the two
|
||||||
|
// content rungs: just below it the whole file is hashed and any byte
|
||||||
|
// difference shows; at the boundary only the gigabyte-spaced samples are
|
||||||
|
// hashed, so a difference outside a sample window is invisible. The
|
||||||
|
// files of each pair match on size, head, and tail, so the scan reads
|
||||||
|
// both for their content hashes.
|
||||||
|
func TestContentRungBoundary(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
|
||||||
|
// A byte that lands outside the single [0, sampleWindow) sample a
|
||||||
|
// sub-gigabyte file has, but well inside the file.
|
||||||
|
const off = 10 * 1024 * 1024
|
||||||
|
|
||||||
|
// Just under the boundary: the whole-file rung sees the poked byte.
|
||||||
|
under := int64(wholeFileMax - 1)
|
||||||
|
underBase := sparseFile(t, dir, "under-base", under)
|
||||||
|
underPoked := sparseFile(t, dir, "under-poked", under)
|
||||||
|
|
||||||
|
pokeAt(t, underPoked, off, []byte{1})
|
||||||
|
|
||||||
|
// At the boundary: only [0, sampleWindow) is sampled, so the poked
|
||||||
|
// byte at off is invisible and the two content hashes match.
|
||||||
|
at := int64(wholeFileMax)
|
||||||
|
atBase := sparseFile(t, dir, "at-base", at)
|
||||||
|
atPoked := sparseFile(t, dir, "at-poked", at)
|
||||||
|
|
||||||
|
pokeAt(t, atPoked, off, []byte{1})
|
||||||
|
|
||||||
|
c := scanContents(t, dir, underBase, underPoked, atBase, atPoked)
|
||||||
|
|
||||||
|
if c[underBase] == c[underPoked] {
|
||||||
|
t.Error("whole-file rung ignored a byte difference below wholeFileMax")
|
||||||
|
}
|
||||||
|
|
||||||
|
if c[atBase] != c[atPoked] {
|
||||||
|
t.Error("sampled rung saw a byte outside every sample window")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestContentRungMultiGigabyte exercises the sampled rung across several
|
||||||
|
// gigabytes using sparse files: a difference inside the third sample
|
||||||
|
// window (at offset 2*sampleStride) changes the hash, while a difference
|
||||||
|
// in the gap after it does not. The three files match on size, head,
|
||||||
|
// and tail, so the scan reads each for its content hash.
|
||||||
|
func TestContentRungMultiGigabyte(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
|
||||||
|
// Three sample windows (offsets 0, 1 GiB, 2 GiB) plus a trailing gap
|
||||||
|
// that no sample covers.
|
||||||
|
size := int64(2*sampleStride + 2*sampleWindow)
|
||||||
|
thirdSample := int64(2 * sampleStride)
|
||||||
|
gap := thirdSample + int64(sampleWindow)
|
||||||
|
|
||||||
|
base := sparseFile(t, dir, "g-base", size)
|
||||||
|
inSample := sparseFile(t, dir, "g-insample", size)
|
||||||
|
inGap := sparseFile(t, dir, "g-ingap", size)
|
||||||
|
|
||||||
|
pokeAt(t, inSample, thirdSample, []byte{1})
|
||||||
|
pokeAt(t, inGap, gap, []byte{1})
|
||||||
|
|
||||||
|
c := scanContents(t, dir, base, inSample, inGap)
|
||||||
|
|
||||||
|
if c[inSample] == c[base] {
|
||||||
|
t.Error("sample at 2 GiB was not read: difference there was invisible")
|
||||||
|
}
|
||||||
|
|
||||||
|
if c[inGap] != c[base] {
|
||||||
|
t.Error("a byte in an unsampled gap changed the content hash")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// sparseFileWithoutMatch writes name in dir as a sparse file of size
|
||||||
|
// bytes, next to another file of that size whose first byte differs. A
|
||||||
|
// scan then reads the file's head and tail, since its size is shared,
|
||||||
|
// but finds no file matching them, so it gets no content hash.
|
||||||
|
func sparseFileWithoutMatch(t *testing.T, dir, name string,
|
||||||
|
size int64,
|
||||||
|
) string {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
p := sparseFile(t, dir, name, size)
|
||||||
|
other := sparseFile(t, dir, name+"-other-head", size)
|
||||||
|
|
||||||
|
pokeAt(t, other, 0, []byte{1})
|
||||||
|
|
||||||
|
return p
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentGate checks that a file of headTailMin or more is read
|
||||||
|
// for its content hash only when its size, head, and tail match another
|
||||||
|
// file's: a same-size pair whose heads differ and one whose tails differ
|
||||||
|
// get no content hash and are not reported, while an identical pair is
|
||||||
|
// read and reported.
|
||||||
|
func TestScanContentGate(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
db := openTestDB(t)
|
||||||
|
|
||||||
|
// Three sizes, so that no pair meets another.
|
||||||
|
headA := sparseFile(t, dir, "head-a", headTailMin)
|
||||||
|
headB := sparseFile(t, dir, "head-b", headTailMin)
|
||||||
|
tailA := sparseFile(t, dir, "tail-a", headTailMin+1)
|
||||||
|
tailB := sparseFile(t, dir, "tail-b", headTailMin+1)
|
||||||
|
same := []string{
|
||||||
|
sparseFile(t, dir, "same-a", headTailMin+2),
|
||||||
|
sparseFile(t, dir, "same-b", headTailMin+2),
|
||||||
|
}
|
||||||
|
|
||||||
|
pokeAt(t, headB, 0, []byte{1})
|
||||||
|
pokeAt(t, tailB, headTailMin, []byte{1}) // its last byte
|
||||||
|
|
||||||
|
syncTree(t, db, dir)
|
||||||
|
|
||||||
|
recs := dbRecords(t, db)
|
||||||
|
for _, p := range []string{headA, headB, tailA, tailB} {
|
||||||
|
r := recordByPath(t, recs, p)
|
||||||
|
if r.head == "" || r.tail == "" || r.content != "" {
|
||||||
|
t.Errorf("%s: head = %q tail = %q content = %q, "+
|
||||||
|
"want head and tail only", p, r.head, r.tail, r.content)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
groups := collectDupeGroups(recs)
|
||||||
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, same) {
|
||||||
|
t.Fatalf("groups = %+v, want only the identical pair %q",
|
||||||
|
groups, same)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentAcrossOperands checks that a stored file gets its
|
||||||
|
// content hash when a later scan of a separate operand brings its
|
||||||
|
// match: tree A's file has a head and tail but no content hash until
|
||||||
|
// tree B, holding an identical file, is scanned.
|
||||||
|
func TestScanContentAcrossOperands(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
a := sparseFileWithoutMatch(t, t.TempDir(), "a", headTailMin)
|
||||||
|
|
||||||
|
syncTree(t, db, filepath.Dir(a))
|
||||||
|
|
||||||
|
if r := recordByPath(t, dbRecords(t, db), a); r.head == "" || r.content != "" {
|
||||||
|
t.Fatalf("after scanning A: %+v, want head and tail only", r)
|
||||||
|
}
|
||||||
|
|
||||||
|
b := sparseFile(t, t.TempDir(), "b", headTailMin)
|
||||||
|
|
||||||
|
syncTree(t, db, filepath.Dir(b))
|
||||||
|
|
||||||
|
recs := dbRecords(t, db)
|
||||||
|
if r := recordByPath(t, recs, a); r.content == "" {
|
||||||
|
t.Fatalf("after scanning B: %+v, want A's file content-hashed", r)
|
||||||
|
}
|
||||||
|
|
||||||
|
want := []string{a, b}
|
||||||
|
slices.Sort(want)
|
||||||
|
|
||||||
|
groups := collectDupeGroups(recs)
|
||||||
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
|
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentWithinOperand checks that a rescan adding a match next
|
||||||
|
// to an unchanged stored file gives the stored file its content hash,
|
||||||
|
// though the hash phase leaves it alone as unchanged.
|
||||||
|
func TestScanContentWithinOperand(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
db := openTestDB(t)
|
||||||
|
stored := sparseFileWithoutMatch(t, dir, "d1", headTailMin)
|
||||||
|
|
||||||
|
syncTree(t, db, dir)
|
||||||
|
|
||||||
|
added := sparseFile(t, dir, "d2", headTailMin)
|
||||||
|
|
||||||
|
st := syncTree(t, db, dir)
|
||||||
|
if st != (scanStats{added: 1, unchanged: 2}) {
|
||||||
|
t.Fatalf("rescan stats = %+v, want 1 added 2 unchanged", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
want := []string{stored, added}
|
||||||
|
|
||||||
|
groups := collectDupeGroups(dbRecords(t, db))
|
||||||
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
|
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentStalePartners checks that a stored file outside the
|
||||||
|
// operand that has vanished, or changed, since it was recorded is not
|
||||||
|
// read, and that its match inside the operand is not read either: the
|
||||||
|
// match has no other partner left, so neither gets a content hash and
|
||||||
|
// no duplicate is reported.
|
||||||
|
func TestScanContentStalePartners(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
dirA := t.TempDir()
|
||||||
|
gone := sparseFileWithoutMatch(t, dirA, "gone", headTailMin)
|
||||||
|
changed := sparseFileWithoutMatch(t, dirA, "changed", headTailMin+1)
|
||||||
|
|
||||||
|
syncTree(t, db, dirA)
|
||||||
|
|
||||||
|
before := dbRecords(t, db)
|
||||||
|
|
||||||
|
err := os.Remove(gone)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
future := time.Now().Add(time.Hour)
|
||||||
|
|
||||||
|
err = os.Chtimes(changed, future, future)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
dirB := t.TempDir()
|
||||||
|
sparseFile(t, dirB, "gone-copy", headTailMin)
|
||||||
|
sparseFile(t, dirB, "changed-copy", headTailMin+1)
|
||||||
|
|
||||||
|
st := syncTree(t, db, dirB)
|
||||||
|
if st != (scanStats{added: 2}) {
|
||||||
|
t.Errorf("stats = %+v, want 2 added and nothing skipped", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
recs := dbRecords(t, db)
|
||||||
|
for _, r := range recs {
|
||||||
|
if r.content != "" {
|
||||||
|
t.Errorf("%s: content = %q, want none: its only match is stale",
|
||||||
|
r.path, r.content)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, old := range before {
|
||||||
|
if r := recordByPath(t, recs, old.path); r != old {
|
||||||
|
t.Errorf("record = %+v, want it left as %+v", r, old)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if groups := collectDupeGroups(recs); len(groups) != 0 {
|
||||||
|
t.Errorf("groups = %+v, want none", groups)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentHashedStalePartners checks that stored matches outside
|
||||||
|
// the operand that already have a content hash are checked like any
|
||||||
|
// other: once one has vanished and the other has changed, a copy of
|
||||||
|
// them scanned in another tree has no match left, so it is not read and
|
||||||
|
// is not reported as their duplicate.
|
||||||
|
func TestScanContentHashedStalePartners(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
dirA := t.TempDir()
|
||||||
|
stored := []string{
|
||||||
|
sparseFile(t, dirA, "changed", headTailMin),
|
||||||
|
sparseFile(t, dirA, "gone", headTailMin),
|
||||||
|
}
|
||||||
|
|
||||||
|
// The two stored files match, so this scan gives both a content
|
||||||
|
// hash.
|
||||||
|
syncTree(t, db, dirA)
|
||||||
|
|
||||||
|
err := os.Remove(stored[1])
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
future := time.Now().Add(time.Hour)
|
||||||
|
|
||||||
|
err = os.Chtimes(stored[0], future, future)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
b := sparseFile(t, t.TempDir(), "copy", headTailMin)
|
||||||
|
|
||||||
|
st := syncTree(t, db, filepath.Dir(b))
|
||||||
|
if st != (scanStats{added: 1}) {
|
||||||
|
t.Errorf("stats = %+v, want 1 added and nothing skipped", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
recs := dbRecords(t, db)
|
||||||
|
if r := recordByPath(t, recs, b); r.content != "" {
|
||||||
|
t.Errorf("copy: content = %q, want none: its only matches are stale",
|
||||||
|
r.content)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The stored records lie outside the operand and are left as they
|
||||||
|
// are, so they still group with each other, but not with the copy.
|
||||||
|
groups := collectDupeGroups(recs)
|
||||||
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) {
|
||||||
|
t.Errorf("groups = %+v, want only the stored pair %q", groups, stored)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentReadFailure checks that a failed content read is
|
||||||
|
// counted as skipped and leaves the record without a content hash, and
|
||||||
|
// that a later scan tries the read again.
|
||||||
|
func TestScanContentReadFailure(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
a := sparseFileWithoutMatch(t, t.TempDir(), "a", headTailMin)
|
||||||
|
|
||||||
|
syncTree(t, db, filepath.Dir(a))
|
||||||
|
|
||||||
|
// lstat still works on the unreadable file, so it passes the check
|
||||||
|
// and fails only when it is read.
|
||||||
|
err := os.Chmod(a, 0)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
dirB := t.TempDir()
|
||||||
|
b := sparseFile(t, dirB, "b", headTailMin)
|
||||||
|
|
||||||
|
st := syncTree(t, db, dirB)
|
||||||
|
if st != (scanStats{added: 1, skipped: 1}) {
|
||||||
|
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
if r := recordByPath(t, dbRecords(t, db), a); r.content != "" {
|
||||||
|
t.Fatalf("unreadable file: %+v, want no content hash", r)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = os.Chmod(a, 0o600)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
st = syncTree(t, db, dirB)
|
||||||
|
if st != (scanStats{unchanged: 1}) {
|
||||||
|
t.Fatalf("rescan stats = %+v, want 1 unchanged", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
want := []string{a, b}
|
||||||
|
slices.Sort(want)
|
||||||
|
|
||||||
|
groups := collectDupeGroups(dbRecords(t, db))
|
||||||
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
|
t.Fatalf("groups = %+v, want the pair %q after the retry",
|
||||||
|
groups, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentCheckError checks that a stored file the content phase
|
||||||
|
// cannot lstat, for a reason other than its being gone, is counted as
|
||||||
|
// skipped and does not count as a match.
|
||||||
|
func TestScanContentCheckError(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
sub := filepath.Join(t.TempDir(), "sub")
|
||||||
|
|
||||||
|
err := os.Mkdir(sub, 0o700)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
sparseFileWithoutMatch(t, sub, "a", headTailMin)
|
||||||
|
syncTree(t, db, sub)
|
||||||
|
|
||||||
|
// Without search permission on its directory, the stored file's
|
||||||
|
// lstat fails with permission denied.
|
||||||
|
err = os.Chmod(sub, 0)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
t.Cleanup(func() {
|
||||||
|
//nolint:gosec // removing the directory needs its search bit back
|
||||||
|
_ = os.Chmod(sub, 0o700)
|
||||||
|
})
|
||||||
|
|
||||||
|
b := sparseFile(t, t.TempDir(), "b", headTailMin)
|
||||||
|
|
||||||
|
st := syncTree(t, db, filepath.Dir(b))
|
||||||
|
if st != (scanStats{added: 1, skipped: 1}) {
|
||||||
|
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
if r := recordByPath(t, dbRecords(t, db), b); r.content != "" {
|
||||||
|
t.Errorf("b: content = %q, want none: its only match could not be "+
|
||||||
|
"checked", r.content)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestScanContentHardlinks checks that the content phase stores the
|
||||||
|
// content hash of a hard-linked file on every one of its links.
|
||||||
|
func TestScanContentHardlinks(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
db := openTestDB(t)
|
||||||
|
a := sparseFile(t, dir, "a", headTailMin)
|
||||||
|
b := filepath.Join(dir, "b")
|
||||||
|
|
||||||
|
err := os.Link(a, b)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
c := sparseFile(t, dir, "copy", headTailMin)
|
||||||
|
|
||||||
|
st := syncTree(t, db, dir)
|
||||||
|
if st != (scanStats{added: 3}) {
|
||||||
|
t.Fatalf("stats = %+v, want 3 added", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
recs := dbRecords(t, db)
|
||||||
|
|
||||||
|
want := recordByPath(t, recs, c).content
|
||||||
|
if want == "" {
|
||||||
|
t.Fatal("the copy has no content hash")
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, p := range []string{a, b} {
|
||||||
|
if got := recordByPath(t, recs, p).content; got != want {
|
||||||
|
t.Errorf("%s: content = %q, want %q", p, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// collectWalk runs a walk over roots and returns the emitted records
|
// collectWalk runs a walk over roots and returns the emitted records
|
||||||
@@ -706,9 +1289,9 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
|||||||
|
|
||||||
recs := dbRecords(t, db)
|
recs := dbRecords(t, db)
|
||||||
for _, r := range recs {
|
for _, r := range recs {
|
||||||
if r.head != "" || r.tail != "" {
|
if r.head != "" || r.tail != "" || r.content != "" {
|
||||||
t.Errorf("%s: head = %q tail = %q, want unhashed",
|
t.Errorf("%s: head = %q tail = %q content = %q, want unhashed",
|
||||||
r.path, r.head, r.tail)
|
r.path, r.head, r.tail, r.content)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -740,23 +1323,36 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
|||||||
func TestTreesUnhashedNeverEqual(t *testing.T) {
|
func TestTreesUnhashedNeverEqual(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// Two trees identical except for unhashed same-name, same-size
|
// Two trees identical except for same-name, same-size files without
|
||||||
// files (possible when the trees were scanned separately) must not
|
// a content hash must not compare equal: their content is unknown.
|
||||||
// compare equal: unhashed content is unknown.
|
// That holds for unhashed files (possible when the trees were
|
||||||
shared := pattern(1, 100)
|
// scanned separately) and for files of headTailMin or more that
|
||||||
recs := []scanRec{
|
// have only a head and tail.
|
||||||
{path: "/x/t1/f1", size: 100, head: hexSum(shared), tail: hexSum(shared)},
|
sum := hexSum(pattern(1, 100))
|
||||||
{path: "/x/t2/f1", size: 100, head: hexSum(shared), tail: hexSum(shared)},
|
shared := []scanRec{
|
||||||
{path: "/x/t1/u", size: 50},
|
{path: "/x/t1/f1", size: 100, head: sum, tail: sum, content: sum},
|
||||||
{path: "/x/t2/u", size: 50},
|
{path: "/x/t2/f1", size: 100, head: sum, tail: sum, content: sum},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := buildHierarchy(recs)
|
cases := map[string][]scanRec{
|
||||||
|
"unhashed": {
|
||||||
|
{path: "/x/t1/u", size: 50},
|
||||||
|
{path: "/x/t2/u", size: 50},
|
||||||
|
},
|
||||||
|
"head and tail only": {
|
||||||
|
{path: "/x/t1/u", size: headTailMin, head: "h", tail: "t"},
|
||||||
|
{path: "/x/t2/u", size: headTailMin, head: "h", tail: "t"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
for name, unknown := range cases {
|
||||||
|
super, dirs := buildHierarchy(append(slices.Clone(shared), unknown...))
|
||||||
super.compute()
|
super.compute()
|
||||||
|
|
||||||
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
|
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
|
||||||
t.Fatalf("tree groups = %d, want 0 (unhashed files differ)",
|
t.Errorf("%s: tree groups = %d, want 0 (the files may differ)",
|
||||||
len(tg))
|
name, len(tg))
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -11,12 +11,6 @@ set -eu
|
|||||||
|
|
||||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||||
|
|
||||||
# yarn provides prettier, which formats Markdown. yarn is a tool, like
|
|
||||||
# node/git/make/go below; the reference that governs formatting output is
|
|
||||||
# prettier, pinned by yarn.lock's integrity hash and installed by
|
|
||||||
# `yarn install --frozen-lockfile`.
|
|
||||||
YARN_VERSION="1.22.22"
|
|
||||||
|
|
||||||
PKGMGR=""
|
PKGMGR=""
|
||||||
SUDO=""
|
SUDO=""
|
||||||
APT_UPDATED=""
|
APT_UPDATED=""
|
||||||
@@ -64,21 +58,6 @@ missing() {
|
|||||||
! command -v "$1" >/dev/null 2>&1
|
! command -v "$1" >/dev/null 2>&1
|
||||||
}
|
}
|
||||||
|
|
||||||
ensure_node() {
|
|
||||||
if ! missing node; then return 0; fi
|
|
||||||
pkg_install nodejs nodejs node nodejs
|
|
||||||
}
|
|
||||||
|
|
||||||
ensure_yarn() {
|
|
||||||
if ! missing yarn; then return 0; fi
|
|
||||||
if ! missing corepack; then
|
|
||||||
corepack enable >/dev/null 2>&1 || true
|
|
||||||
corepack prepare "yarn@$YARN_VERSION" --activate
|
|
||||||
else
|
|
||||||
pkg_install yarn yarn yarn yarn
|
|
||||||
fi
|
|
||||||
}
|
|
||||||
|
|
||||||
main() {
|
main() {
|
||||||
cd "$ROOT"
|
cd "$ROOT"
|
||||||
|
|
||||||
@@ -92,16 +71,6 @@ main() {
|
|||||||
if missing make; then pkg_install gnumake make make make; fi
|
if missing make; then pkg_install gnumake make make make; fi
|
||||||
if missing go; then pkg_install go golang go go; fi
|
if missing go; then pkg_install go golang go go; fi
|
||||||
|
|
||||||
# node runs prettier and is an unpinned host tool for the same reason
|
|
||||||
# git/make/go are: it comes from the host package manager, whatever
|
|
||||||
# version it ships. It is not installed via nvm the way the canonical
|
|
||||||
# template does, because nvm's prebuilt node is glibc-linked and does
|
|
||||||
# not run on this repo's musl/Alpine build image. prettier — the tool
|
|
||||||
# whose version affects formatting output — is pinned by yarn.lock.
|
|
||||||
ensure_node
|
|
||||||
ensure_yarn
|
|
||||||
yarn install --frozen-lockfile
|
|
||||||
|
|
||||||
# Linting runs via docker only (script/lint), so docker is a lint
|
# Linting runs via docker only (script/lint), so docker is a lint
|
||||||
# prerequisite rather than something bootstrap installs. Warn, do
|
# prerequisite rather than something bootstrap installs. Warn, do
|
||||||
# not fail: everything except `make lint` — and, through it,
|
# not fail: everything except `make lint` — and, through it,
|
||||||
|
|||||||
+4
-4
@@ -16,10 +16,10 @@
|
|||||||
# implies the repo is green.
|
# implies the repo is green.
|
||||||
#
|
#
|
||||||
# That implication holds only because of CHECK_EPOCH. A COPY layer is
|
# That implication holds only because of CHECK_EPOCH. A COPY layer is
|
||||||
# invalidated by changed content, and a merge commit's tree is
|
# invalidated only by changed content, and a rebuild of an unchanged
|
||||||
# byte-identical to the branch head it merges, so without a fresh value
|
# checkout sends the same content, so without a fresh value here Docker
|
||||||
# here Docker serves the gate layers from cache and the build reports a
|
# serves the gate layers from cache and the build reports a green it
|
||||||
# green it never earned. Passing the current epoch invalidates the gate
|
# never earned. Passing the current epoch invalidates the gate
|
||||||
# layers on every run while leaving the pinned base images and
|
# layers on every run while leaving the pinned base images and
|
||||||
# go mod download cached; see the Dockerfile for the placement.
|
# go mod download cached; see the Dockerfile for the placement.
|
||||||
set -eu
|
set -eu
|
||||||
|
|||||||
+1
-12
@@ -1,23 +1,12 @@
|
|||||||
#!/bin/sh
|
#!/bin/sh
|
||||||
# script/fmt: format all files (writes). gofmt for Go, prettier for
|
# script/fmt: format all files (writes).
|
||||||
# Markdown. prettier is the pinned devDependency in package.json/
|
|
||||||
# yarn.lock; script/bootstrap installs it (see run_prettier).
|
|
||||||
set -eu
|
set -eu
|
||||||
|
|
||||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||||
|
|
||||||
run_prettier() {
|
|
||||||
if ! command -v yarn >/dev/null 2>&1; then
|
|
||||||
echo "fmt: yarn not found; run script/bootstrap first" >&2
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
yarn run prettier "$@"
|
|
||||||
}
|
|
||||||
|
|
||||||
main() {
|
main() {
|
||||||
cd "$ROOT"
|
cd "$ROOT"
|
||||||
gofmt -s -w .
|
gofmt -s -w .
|
||||||
run_prettier --write '**/*.md' --tab-width 4 --prose-wrap always
|
|
||||||
}
|
}
|
||||||
|
|
||||||
main "$@"
|
main "$@"
|
||||||
|
|||||||
+2
-21
@@ -1,37 +1,18 @@
|
|||||||
#!/bin/sh
|
#!/bin/sh
|
||||||
# script/fmt-check: check formatting (read-only). Same scope as
|
# script/fmt-check: check formatting (read-only). Same scope as
|
||||||
# script/fmt: gofmt for Go, prettier for Markdown. Both run every time
|
# script/fmt, but fails instead of writing.
|
||||||
# and each reports independently, so a failure names which formatter is
|
|
||||||
# unhappy; the script exits non-zero if either found unformatted files.
|
|
||||||
set -eu
|
set -eu
|
||||||
|
|
||||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||||
|
|
||||||
run_prettier() {
|
|
||||||
if ! command -v yarn >/dev/null 2>&1; then
|
|
||||||
echo "fmt-check: yarn not found; run script/bootstrap first" >&2
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
yarn run prettier "$@"
|
|
||||||
}
|
|
||||||
|
|
||||||
main() {
|
main() {
|
||||||
cd "$ROOT"
|
cd "$ROOT"
|
||||||
rc=0
|
|
||||||
|
|
||||||
files="$(gofmt -s -l .)"
|
files="$(gofmt -s -l .)"
|
||||||
if [ -n "$files" ]; then
|
if [ -n "$files" ]; then
|
||||||
echo "gofmt: files not formatted:" >&2
|
echo "gofmt: files not formatted:" >&2
|
||||||
echo "$files" >&2
|
echo "$files" >&2
|
||||||
rc=1
|
exit 1
|
||||||
fi
|
fi
|
||||||
|
|
||||||
if ! run_prettier --check '**/*.md' --tab-width 4 --prose-wrap always; then
|
|
||||||
echo "prettier: Markdown not formatted; run make fmt" >&2
|
|
||||||
rc=1
|
|
||||||
fi
|
|
||||||
|
|
||||||
exit "$rc"
|
|
||||||
}
|
}
|
||||||
|
|
||||||
main "$@"
|
main "$@"
|
||||||
|
|||||||
@@ -16,6 +16,7 @@ type fileSig struct {
|
|||||||
size int64
|
size int64
|
||||||
head string
|
head string
|
||||||
tail string
|
tail string
|
||||||
|
content string
|
||||||
}
|
}
|
||||||
|
|
||||||
// treeNode is one directory reconstructed from the scan stream.
|
// treeNode is one directory reconstructed from the scan stream.
|
||||||
@@ -122,14 +123,16 @@ func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
|||||||
node.files = make(map[string]fileSig)
|
node.files = make(map[string]fileSig)
|
||||||
}
|
}
|
||||||
|
|
||||||
sig := fileSig{size: r.size, head: r.head, tail: r.tail}
|
sig := fileSig{
|
||||||
|
size: r.size, head: r.head, tail: r.tail, content: r.content,
|
||||||
|
}
|
||||||
|
|
||||||
// An unhashed record (its size was unique when last scanned)
|
// A record without a content hash has unknown content (README
|
||||||
// has unknown content: give it a signature no other file can
|
// "Database"): give it a signature no other file can share, so
|
||||||
// share, so trees containing it never compare equal. Real
|
// trees containing it never compare equal. Real hashes are
|
||||||
// heads are hex, so the NUL-prefixed form cannot collide.
|
// hex, so the NUL-prefixed form cannot collide.
|
||||||
if sig.head == "" {
|
if sig.content == "" {
|
||||||
sig.head = "unhashed\x00" + r.path
|
sig.content = "unhashed\x00" + r.path
|
||||||
}
|
}
|
||||||
|
|
||||||
node.files[comps[len(comps)-1]] = sig
|
node.files[comps[len(comps)-1]] = sig
|
||||||
@@ -188,7 +191,7 @@ func (n *treeNode) compute() {
|
|||||||
for name, sig := range n.files {
|
for name, sig := range n.files {
|
||||||
entries = append(entries,
|
entries = append(entries,
|
||||||
"f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+
|
"f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+
|
||||||
"\x00"+sig.head+"\x00"+sig.tail)
|
"\x00"+sig.head+"\x00"+sig.tail+"\x00"+sig.content)
|
||||||
n.fileCount++
|
n.fileCount++
|
||||||
n.totalSize += sig.size
|
n.totalSize += sig.size
|
||||||
}
|
}
|
||||||
|
|||||||
+16
-13
@@ -9,20 +9,23 @@ import (
|
|||||||
const (
|
const (
|
||||||
f1Head = "f1h"
|
f1Head = "f1h"
|
||||||
f1Tail = "f1t"
|
f1Tail = "f1t"
|
||||||
|
f1Content = "f1c"
|
||||||
f2Head = "f2h"
|
f2Head = "f2h"
|
||||||
f2Tail = "f2t"
|
f2Tail = "f2t"
|
||||||
|
f2Content = "f2c"
|
||||||
)
|
)
|
||||||
|
|
||||||
// smokeTreeRecs mirrors the README smoke-test tree layout: /d/t1 and
|
// smokeTreeRecs mirrors the README smoke-test tree layout: /d/t1 and
|
||||||
// /d/t2 are identical, /d/t3 differs from them only by one filename.
|
// /d/t2 are identical, /d/t3 differs from them only by one filename.
|
||||||
func smokeTreeRecs() []scanRec {
|
func smokeTreeRecs() []scanRec {
|
||||||
return []scanRec{
|
return []scanRec{
|
||||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t1/f1"},
|
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t1/f1"},
|
||||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t1/sub/f2"},
|
{size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t1/sub/f2"},
|
||||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t2/f1"},
|
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t2/f1"},
|
||||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t2/sub/f2"},
|
{size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t2/sub/f2"},
|
||||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t3/f1"},
|
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t3/f1"},
|
||||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t3/sub/f2renamed"},
|
{size: 100, head: f2Head, tail: f2Tail, content: f2Content,
|
||||||
|
path: "/d/t3/sub/f2renamed"},
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -114,8 +117,8 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
|
|||||||
const sharedTail = "same"
|
const sharedTail = "same"
|
||||||
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 10, head: sharedTail, tail: sharedTail, path: "/r/a/f"},
|
{size: 10, head: sharedTail, tail: sharedTail, content: "c", path: "/r/a/f"},
|
||||||
{size: 10, head: "DIFF", tail: sharedTail, path: "/r/b/f"},
|
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := buildHierarchy(recs)
|
super, dirs := buildHierarchy(recs)
|
||||||
@@ -181,8 +184,8 @@ func TestCollectTreeGroupsSiblings(t *testing.T) {
|
|||||||
// Identical sibling dirs share a parent, so their group cannot be
|
// Identical sibling dirs share a parent, so their group cannot be
|
||||||
// implied by a parent group and must be reported.
|
// implied by a parent group and must be reported.
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 10, head: "h", tail: "t", path: "/p/x1/f"},
|
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x1/f"},
|
||||||
{size: 10, head: "h", tail: "t", path: "/p/x2/f"},
|
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := buildHierarchy(recs)
|
super, dirs := buildHierarchy(recs)
|
||||||
@@ -203,9 +206,9 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) {
|
|||||||
// extra file, so the parents' digests differ and the x group must
|
// extra file, so the parents' digests differ and the x group must
|
||||||
// be reported.
|
// be reported.
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 10, head: "h", tail: "t", path: "/p/a/x/f"},
|
{size: 10, head: "h", tail: "t", content: "c", path: "/p/a/x/f"},
|
||||||
{size: 99, head: "e", tail: "e", path: "/p/a/extra"},
|
{size: 99, head: "e", tail: "e", content: "e", path: "/p/a/extra"},
|
||||||
{size: 10, head: "h", tail: "t", path: "/q/b/x/f"},
|
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := buildHierarchy(recs)
|
super, dirs := buildHierarchy(recs)
|
||||||
|
|||||||
@@ -1,8 +0,0 @@
|
|||||||
# THIS IS AN AUTOGENERATED FILE. DO NOT EDIT THIS FILE DIRECTLY.
|
|
||||||
# yarn lockfile v1
|
|
||||||
|
|
||||||
|
|
||||||
prettier@3.8.1:
|
|
||||||
version "3.8.1"
|
|
||||||
resolved "https://registry.yarnpkg.com/prettier/-/prettier-3.8.1.tgz#edf48977cf991558f4fcbd8a3ba6015ba2a3a173"
|
|
||||||
integrity sha512-UOnG6LftzbdaHZcKoPFtOcCKztrQ57WkHDeRD9t/PTQtmT0NHSeWWepj6pS0z/N7+08BHFDQVUrfmfMRcZwbMg==
|
|
||||||
Reference in New Issue
Block a user