1 Commits
Author SHA1 Message Date
clawbot 1b0b298e68 Print progress at once off a terminal, keep warnings out of redraws (closes #13)
check / check (push) Successful in 1m35s
When stderr is not a terminal, each phase prints its zero-state line
as it starts instead of after its first item. stderrIsTTY uses
term.IsTerminal from golang.org/x/term, now a direct dependency, so
/dev/null is no longer taken for a terminal.

A spinner keeps the library's background redraw, so its count and
elapsed time stay current while a phase waits for its next item. A
warning printed during a spinner phase goes through the bar
(progressbar.Bprintln), which prints it before its next redraw instead
of racing it. Bars with a total have no background redraw and still
print warnings directly.

Model: opus-5-5
2026-10-04 02:35:42 +00:00
31 changed files with 1704 additions and 3177 deletions
+2 -66
View File
@@ -10,20 +10,14 @@ run:
linters: linters:
default: all default: all
enable:
# Successor to the deprecated gomodguard. Named explicitly, rather than
# left to `default: all`, because it carries the module policy below.
- gomodguard_v2
disable: disable:
# Genuinely incompatible with project patterns # Genuinely incompatible with project patterns
- exhaustruct # Requires all struct fields - exhaustruct # Requires all struct fields
- depguard # Dependency allow/block lists
- godot # Requires comments to end with periods - godot # Requires comments to end with periods
- wsl # Deprecated, replaced by wsl_v5
- wrapcheck # Too verbose for internal packages - wrapcheck # Too verbose for internal packages
- varnamelen # Short names like db, id are idiomatic Go - varnamelen # Short names like db, id are idiomatic Go
# Deprecated: the warning is attached to the old name, so it is
# silenced by disabling that name, not by enabling the successor.
- wsl # Deprecated, replaced by wsl_v5
- gomodguard # Deprecated, replaced by gomodguard_v2
settings: settings:
lll: lll:
line-length: 88 line-length: 88
@@ -34,64 +28,6 @@ linters:
max-complexity: 15 max-complexity: 15
dupl: dupl:
threshold: 100 threshold: 100
depguard:
# Test-support code must not be compiled into the shipped binary. A
# test-support package exists to hand a test privileges the program
# itself must never have, so a file that is not a test must not import
# one. Test files, and the files inside a package whose directory name
# ends in `test`, are where that code belongs, and are exempt.
#
# The deny list below is the one part of this file a repository is
# expected to extend, and the only part it may. depguard matches an
# import path against a list of prefixes, so it cannot be told "any path
# whose last segment ends in test"; a repository's own test-support
# packages have to be named here one at a time, by full import path,
# under a module path that differs from repository to repository. Add
# them; change nothing else.
rules:
test-support:
list-mode: lax
files:
- "$all"
- "!$test"
- "!**/*test/**"
deny:
- pkg: net/http/httptest
desc: >-
Test-support code belongs in test files and in packages whose
directory name ends in test, not in the shipped binary.
# Only decisions already recorded in the Go package defaults are
# listed here. Every entry matches the module path exactly.
gomodguard_v2:
blocked:
- module: github.com/rs/zerolog
recommendations:
- log/slog
reason: "Structured logging is stdlib log/slog."
# One entry per pre-fork module path, because the later releases
# are separate paths. A prefix match would be shorter but would
# also reach github.com/go-redis/redismock, the test double for
# the successor these entries recommend.
- module: github.com/go-redis/redis
recommendations:
- github.com/redis/go-redis/v9
reason: "Pre-fork module; use the maintained go-redis v9."
- module: github.com/go-redis/redis/v7
recommendations:
- github.com/redis/go-redis/v9
reason: "Pre-fork module; use the maintained go-redis v9."
- module: github.com/go-redis/redis/v8
recommendations:
- github.com/redis/go-redis/v9
reason: "Pre-fork module; use the maintained go-redis v9."
- module: github.com/sergi/go-diff
recommendations:
- github.com/aymanbagabas/go-udiff
reason: "No unified diff output; use go-udiff."
- module: github.com/hexops/gotextdiff
recommendations:
- github.com/aymanbagabas/go-udiff
reason: "Unmaintained fork; use go-udiff."
issues: issues:
max-issues-per-linter: 0 max-issues-per-linter: 0
-2
View File
@@ -1,2 +0,0 @@
node_modules/
yarn.lock
-4
View File
@@ -1,4 +0,0 @@
{
"tabWidth": 4,
"proseWrap": "always"
}
+59 -98
View File
@@ -6,36 +6,34 @@ COPY go.mod go.sum ./
RUN go mod download RUN go mod download
COPY . . COPY . .
# Cache-buster for the gate layers, and only for them: on an unchanged # Cache-buster for the gate layers, and only for them. Docker
# tree Docker would serve the gates below from cache and the build would # invalidates COPY only when the copied content changes, so on an
# exit 0 having run nothing. script/cibuild and script/docker pass a # unchanged tree the gates below would be served from cache and the
# fresh CHECK_EPOCH; a build without one, such as a bare # build would exit 0 having run nothing. script/cibuild and
# `docker build .`, fails at the check right after the ARG. # script/docker pass a fresh CHECK_EPOCH on every invocation.
# #
# ARG is per-stage, so the markdown and build stages declare it again. # Two properties this depends on. ARG is per-stage, so the build stage
# Each gate RUN must reference the value: BuildKit hashes the expanded # below declares it again; one declaration here would leave that
# command, so a declared but unreferenced ARG invalidates nothing. Keep # stage's gate cacheable. And each gate RUN must reference the value,
# it below the dependency layers so they stay cached. # because BuildKit hashes the expanded command: a declared but
# unreferenced ARG invalidates nothing.
#
# It sits below the dependency layers deliberately. Everything above it
# (the pinned base image, go mod download) keeps its cache; only the
# gates go cold.
ARG CHECK_EPOCH ARG CHECK_EPOCH
RUN if [ -z "${CHECK_EPOCH}" ]; then \
echo "CHECK_EPOCH is unset; build via script/cibuild or script/docker" >&2; \
exit 1; \
fi
# These gates call the tools directly, not through `make lint` or # The linter is invoked directly here, not through `make lint`. That
# `make fmt-check`: both run docker, which cannot run inside a docker # target now runs `docker build -f Dockerfile.lint`, and a docker build
# build. This step is the gofmt half of `make fmt-check`; the markdown # cannot run a docker build: routing the gate through make would mean
# stage is its prettier half. gofmt's output is assigned to a variable # nesting docker inside this image. Same reason `make check` is gone
# first so that its own exit status, as when it cannot parse a file, # from the build stage below. `make fmt-check` stays as it is — it is a
# still fails the step. # gate, not the aggregate, and it shells out to nothing.
RUN echo "gate gofmt, epoch ${CHECK_EPOCH}" && \ RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
files="$(gofmt -s -l .)" && \
if [ -n "$files" ]; then \
echo "gofmt: files not formatted:" >&2; echo "$files" >&2; exit 1; \
fi
# Fails the build when the FROM above and the one in Dockerfile.lint pin # The FROM above and the one in Dockerfile.lint pin the same linter
# different linter images. # twice, and nothing else keeps them in sync; this fails the build when
# they disagree. See the script for why it restates neither pin.
RUN echo "gate lint-image-pin, epoch ${CHECK_EPOCH}" && \ RUN echo "gate lint-image-pin, epoch ${CHECK_EPOCH}" && \
script/verify-lint-image-pin script/verify-lint-image-pin
@@ -48,104 +46,67 @@ RUN echo "gate config verify, epoch ${CHECK_EPOCH}" && \
RUN echo "gate lint, epoch ${CHECK_EPOCH}" && \ RUN echo "gate lint, epoch ${CHECK_EPOCH}" && \
golangci-lint run --config .golangci.yml ./... golangci-lint run --config .golangci.yml ./...
# Prettier stage: the prettier that formats this repository's Markdown,
# never installed on a host. script/fmt and script/fmt-check build this
# stage alone and run it with the repository mounted on /src. prettier
# is installed in /tools so that the repository, mounted or copied onto
# /src, cannot hide it.
# node:22-alpine, 2026-02-22
FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS prettier
WORKDIR /tools
# yarn.lock pins prettier by hash, and --frozen-lockfile fails rather
# than install anything yarn.lock does not name.
COPY package.json yarn.lock ./
RUN yarn install --frozen-lockfile
ENV PATH=/tools/node_modules/.bin:$PATH
WORKDIR /src
# Markdown stage: the Markdown half of `make fmt-check`, as a gate.
FROM prettier AS markdown
COPY . .
# Second per-stage declaration of the gate cache-buster and its check;
# see the lint stage above.
ARG CHECK_EPOCH
RUN if [ -z "${CHECK_EPOCH}" ]; then \
echo "CHECK_EPOCH is unset; build via script/cibuild or script/docker" >&2; \
exit 1; \
fi
RUN echo "gate prettier, epoch ${CHECK_EPOCH}" && \
prettier --check '**/*.md' --tab-width 4 --prose-wrap always
# Build stage # Build stage
# golang:1.25-alpine, 2026-07-23 # golang:1.25-alpine, 2026-07-23
FROM golang@sha256:56961d79ea8129efddcc0b8643fd8a5416b4e6228cfd477e3fd61deb2672c587 AS builder FROM golang@sha256:56961d79ea8129efddcc0b8643fd8a5416b4e6228cfd477e3fd61deb2672c587 AS builder
# We never build or run as root. Create an unprivileged user and point # We never build or run as root. Create an unprivileged user and point
# HOME and the build cache at its home so go build and go test can write # HOME and the Go caches at its home so go build and go test can write
# it when we drop to it below. # their caches when we drop to it below. $GOPATH/bin is deliberately not
# # on PATH: script/bootstrap no longer `go install`s anything (the linter
# The module cache stays at the base image's default /go/pkg/mod and # runs from a pinned image, never from a host install), so nothing lands
# belongs to root: script/bootstrap fills it as root. Do not move it # there and adding it would only widen what this image resolves.
# into the home and hand it over with `chown -R`: that walks every file
# in it, which took from about 80 s to over ten minutes on a shared
# host, depending on load.
RUN adduser -D -u 1000 builder RUN adduser -D -u 1000 builder
ENV HOME=/home/builder ENV HOME=/home/builder
ENV GOPATH=/home/builder/go ENV GOPATH=/home/builder/go
ENV GOMODCACHE=/go/pkg/mod
ENV GOCACHE=/home/builder/.cache/go-build ENV GOCACHE=/home/builder/.cache/go-build
WORKDIR /src WORKDIR /src
# No-op file copies whose only purpose is the build-graph edge: they # No-op file copy whose only purpose is the build-graph edge: it is what
# make this stage depend on the lint and markdown stages, so BuildKit # makes this stage depend on the lint stage, and so what forces BuildKit
# finishes those gates before compilation and tests start. Remove one # to finish fmt-check, the pin guard and lint before compilation and
# and the build silently stops gating on that stage and still exits 0. # tests start. Remove it and the fail-fast design dies silently — the
# build stops gating on lint and still exits 0. It replaces a copy of
# the linter binary itself, which is no longer wanted here: nothing in
# this stage runs the linter, because `make lint` is now a docker build
# and a docker build cannot run inside one.
COPY --from=lint /src/go.sum /dev/null COPY --from=lint /src/go.sum /dev/null
COPY --from=markdown /src/go.sum /dev/null
# Install development prerequisites the same way a developer does. Only # Install development prerequisites the same way a developer does,
# script/ and the dependency manifests are copied first, so this layer # rather than duplicating the installs inline. Only script/ and the
# stays cached until they change. Bootstrap ends in `go mod download`. # dependency manifests are copied first, nothing else, so this layer
# stays cached until the scripts or the dependencies change — bootstrap
# ends in `go mod download`, which is why there is no separate
# invocation of it here.
COPY script/ script/ COPY script/ script/
COPY go.mod go.sum ./ COPY go.mod go.sum ./
RUN script/bootstrap RUN script/bootstrap
# Hand builder only what it writes to, without walking the module cache. COPY . .
# This layer stays cached with bootstrap.
# - /src itself: make build writes the binary into it, and git refuses
# a repository whose top directory belongs to another user.
# - the module cache's cache/download directory itself, not what is in
# it: Go only reads the downloaded modules, but make build saves its
# lookup of this module's own version from git there, in a new
# directory named after the module path.
# - builder's home: the go commands bootstrap ran as root left Go's
# telemetry files there, a few small files.
RUN chown builder:builder /src /go/pkg/mod/cache/download && \
chown -R builder:builder /home/builder
# The sources are handed to builder as they are copied, so no layer has # Hand the sources and caches to the unprivileged user, then drop root
# to walk them. Then drop root before running any checks or builds. # before running any checks or builds.
COPY --chown=builder:builder . . RUN chown -R builder:builder /src /home/builder
USER builder USER builder
# Fail the build unless the branch is green. Runs as non-root: root # Fail the build unless the branch is green. Runs as non-root so the
# would bypass the chmod(0) the permission-denied tests rely on. # permission-denied test paths are exercised legitimately (root would
# bypass the chmod(0) the tests rely on).
# #
# The gate is `make test`, not `make check`, which runs docker; lint and # The gates are the individual targets, not `make check`: that aggregate
# the format checks ran in the lint and markdown stages above. `make`, # runs `script/lint`, which is now a docker build, and nothing inside an
# not the script directly, because the Makefile's # image build may shell out to docker. Lint is not skipped by this — it
# ran in the lint stage above, which this stage's COPY --from makes a
# prerequisite. `make`, not the scripts directly, because the Makefile's
# `export CGO_ENABLED = 0` applies only to what it invokes. # `export CGO_ENABLED = 0` applies only to what it invokes.
# #
# Third per-stage declaration of the gate cache-buster and its check; # Second per-stage declaration of the gate cache-buster; see the lint
# see the lint stage above. It is placed after USER so the drop to the # stage above for why one is not enough. It is placed after USER so the
# unprivileged user still happens before the checks run. # drop to the unprivileged user still happens before the checks run.
ARG CHECK_EPOCH ARG CHECK_EPOCH
RUN if [ -z "${CHECK_EPOCH}" ]; then \
echo "CHECK_EPOCH is unset; build via script/cibuild or script/docker" >&2; \
exit 1; \
fi
RUN echo "gate test, epoch ${CHECK_EPOCH}" && make test RUN echo "gate test, epoch ${CHECK_EPOCH}" && make test
RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
# The version stamped into the binary: the VERSION build argument when # The version stamped into the binary: the VERSION build argument when
# one is given, otherwise `git describe --tags --always` of the .git in # one is given, otherwise `git describe --tags --always` of the .git in
+35 -21
View File
@@ -1,12 +1,14 @@
# Lint-only image, built by script/lint: the repo is copied into the # Lint-only image: this is how the linter runs, everywhere. The repo is
# pinned golangci-lint image and the linter runs as a build step, so a # COPYed into the pinned golangci-lint image and the linter runs as a
# successful build is a clean lint. No bind mount, so it works when the # build step, so a successful build IS a clean lint. golangci-lint is
# docker daemon is remote. # never installed on a host — one toolchain, pinned by digest, identical
# on a laptop and in CI — and this works even when the docker daemon is
# remote and bind mounts are impossible.
# #
# It is separate from the main Dockerfile's lint stage because # script/lint builds this file. It is a separate image from the lint
# script/lint must not depend on the rest of that build; the two FROM # stage of the main Dockerfile because script/lint must not depend on
# lines are kept identical by script/verify-lint-image-pin, run as a # the rest of that build; the two FROM lines are kept identical by
# gate below. # script/verify-lint-image-pin, run as a gate below.
# golangci/golangci-lint:v2.12.2, 2026-08-07 # golangci/golangci-lint:v2.12.2, 2026-08-07
FROM golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240 FROM golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240
@@ -18,26 +20,38 @@ RUN go mod download
COPY . . COPY . .
# Cache-buster for the gate layers, and only for them; caching of the # Cache-buster for the gate layers, and only for them. Caching of the
# lint run is waived by ruling. On an unchanged tree the gates below # lint run is waived by ruling: COPY is invalidated only by changed
# would be served from cache and this build would exit 0 in under a # content, so on an unchanged tree the gates below would be served from
# second having run no linter. script/lint passes a fresh value on every # cache and this build would exit 0 in under a second having run no
# linter at all. That exact false green has bitten this repo twice
# already (#32, #39). script/lint passes a fresh value on every
# invocation. # invocation.
# #
# Each gate RUN must reference the value: BuildKit hashes the expanded # Each gate RUN must reference the value, because BuildKit hashes the
# command, so a declared but unreferenced ARG invalidates nothing. Keep # expanded command and not the ARG declaration: a declared but
# it below the dependency layers so they stay cached. # unreferenced ARG invalidates nothing. The ARG sits below the
# dependency layers deliberately — everything above it keeps its cache,
# only the gates go cold.
ARG CHECK_EPOCH ARG CHECK_EPOCH
# Fails the build when the FROM above and the main Dockerfile's lint # The linter version is pinned in two places, here and in the main
# stage pin different linter images. # Dockerfile's lint stage. Nothing else keeps them in sync, so a
# half-applied bump is a build failure; see the script.
RUN echo "gate lint-image-pin, epoch ${CHECK_EPOCH}" && \ RUN echo "gate lint-image-pin, epoch ${CHECK_EPOCH}" && \
script/verify-lint-image-pin script/verify-lint-image-pin
# Validates .golangci.yml against golangci-lint's JSON schema, which the # Validates .golangci.yml against golangci-lint's JSON schema. The
# pinned binary embeds: measured under `--network none`, it passes a # concern about this step was that it fetches that schema over a live,
# valid config and rejects an invalid one. No gate step makes a network # unpinned HTTPS call; measured on the pinned image, it does not. The
# call, but `go mod download` above needs the network on a cold cache. # binary carries the schema for its own version, so under
# `--network none` this both passes on a valid config and still rejects
# an invalid one with the jsonschema error. That holds for the gate
# steps generally — none of them makes a network call — but not for
# this build as a whole: `go mod download` above needs the network on a
# cold cache, and under `--network none` a first build fails there
# before reaching any gate. That layer stays cached, so only a warm
# cache lints offline, until go.mod or go.sum changes.
RUN echo "gate config verify, epoch ${CHECK_EPOCH}" && \ RUN echo "gate config verify, epoch ${CHECK_EPOCH}" && \
golangci-lint config verify --config .golangci.yml golangci-lint config verify --config .golangci.yml
+1 -4
View File
@@ -6,7 +6,7 @@ BINARY := sfdupes
VERSION := $(shell git describe --tags --always --dirty 2>/dev/null || echo dev) VERSION := $(shell git describe --tags --always --dirty 2>/dev/null || echo dev)
LDFLAGS := -X main.Version=$(VERSION) LDFLAGS := -X main.Version=$(VERSION)
.PHONY: sfdupes build bootstrap setup test test-race lint fmt fmt-check check docker hooks clean .PHONY: sfdupes build bootstrap setup test lint fmt fmt-check check docker hooks clean
# Standard targets are thin shims; the implementations live in script/ # Standard targets are thin shims; the implementations live in script/
# per the scripts-to-rule-them-all pattern. # per the scripts-to-rule-them-all pattern.
@@ -27,9 +27,6 @@ setup:
test: test:
@script/test @script/test
test-race:
@script/test-race
lint: lint:
@script/lint @script/lint
+626 -737
View File
File diff suppressed because it is too large Load Diff
+430 -344
View File
@@ -1,402 +1,488 @@
# Workflow # Workflow
- take an issue from the `1.0.0` milestone on the tracker; work not yet on the - take an issue from the `1.0.0` milestone on the tracker; work not
tracker gets filed as an issue first yet on the tracker gets filed as an issue first
- branch from `next` - branch (from `main`)
- do the work, with tests, in small focused commits - do the work, with tests, in small focused commits
- record it at the top of Completed Steps (`TODO.md` changes in the same commit - record it at the top of Completed Steps (`TODO.md` changes in the
as the work) same commit as the work)
- push the branch and open a PR against `next` whose title ends with - push the branch and open a PR whose title ends with
` (closes #N)` ` (closes #N)`
- an independent review gates each merge to `next`; every finding is addressed - an independent review gates the merge; every finding is addressed
or explicitly rebutted on the PR or explicitly rebutted on the PR
- only the owner merges `next` to `main` - merge to `main` once the review passes
# Status # Status
- pre-1.0 - pre-1.0
- the Gitea tracker is authoritative for the pre-1.0 backlog: the open issues - the Gitea tracker is authoritative for the pre-1.0 backlog: the
under the `1.0.0` milestone are what remains before the tag, and this file open issues under the `1.0.0` milestone are what remains before
records history and process, not the queue the tag, and this file records history and process, not the queue
# Next Step # Next Step
- take the next issue from the `1.0.0` milestone on the tracker: - take the next issue from the `1.0.0` milestone on the tracker:
https://git.eeqj.de/sneak/sfdupes/milestone/17 — the milestone is the source https://git.eeqj.de/sneak/sfdupes/milestone/17 — the milestone is
of truth for what is left before 1.0.0. Individual issues are deliberately not the source of truth for what is left before 1.0.0. Individual
restated here; a copy in this file drifts out of date the moment the tracker issues are deliberately not restated here; a copy in this file
moves drifts out of date the moment the tracker moves
# Completed Steps # Completed Steps
- cut the narration from `TODO.md` Completed Steps and from the comments in - progress prints at once on a non-terminal, uses a real terminal test,
`script/` and both Dockerfiles; §Workflow now branches from and merges to and prints warnings through a spinner instead of racing its redraw
`next` (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/49) (2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/13)
- `make test-race` runs the test suite under the race detector in a cgo-enabled - warn about and skip symlink, socket, FIFO, device and `.zfs`
container, outside `make check` (2026-10-04, operands, keeping the records beneath them (2026-10-03,
https://git.eeqj.de/sneak/sfdupes/issues/18)
- a bare `docker build .` fails with a message naming `script/cibuild` and
`script/docker` instead of serving the gates from cache (2026-10-04,
https://git.eeqj.de/sneak/sfdupes/issues/39)
- `make fmt` and `make fmt-check` run prettier over all Markdown, in Docker, and
CI checks it; all Markdown reformatted (2026-10-04,
https://git.eeqj.de/sneak/sfdupes/issues/19)
- `script/lint` writes no image, so a run no longer leaves an untagged one
behind (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/48)
- tests cover a missing database, `scan` keeping stdout empty, its skip warning,
the `report` and `trees` summary lines, and every subcommand going through
`runE` (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/16)
- `.golangci.yml` replaced with the current canonical copy, which uses
`gomodguard_v2`, so lint no longer prints a deprecation warning (2026-10-04,
https://git.eeqj.de/sneak/sfdupes/issues/26)
- a test fails when either `hashWorker` cancellation check in `scan.go` is
removed (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/83)
- a database path holding `?`, `#` or `%` opens exactly the file it names
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/55)
- `scan` rejects `--workers` below 1 as a usage error instead of running
single-threaded (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/10)
- a test fails when either walk cancellation check in `scan.go` is removed
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/81)
- test that `scan` refuses a database with another schema version (2026-10-04,
https://git.eeqj.de/sneak/sfdupes/issues/64)
- correct four inaccurate comments in `cancel_test.go` and rename
`walkCancelInFlightDirs` to `walkCancelInFlightFiles` (2026-10-04,
https://git.eeqj.de/sneak/sfdupes/issues/33)
- test the `-x` filesystem-boundary rules in `subdirJob` (2026-10-04,
https://git.eeqj.de/sneak/sfdupes/issues/17)
- `scan` creates the schema in one transaction; a version-0 database with a
`files` table is refused with a clear schema-version error (2026-10-04,
https://git.eeqj.de/sneak/sfdupes/issues/11)
- README documents install, Docker, a daily cron scan and how to read and check
the reports (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/54)
- the `Dockerfile` build stage keeps the Go module cache out of `builder`'s home
and copies the sources with `--chown`, so no `chown -R` walks them
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/43)
- `--version` prints `sfdupes VERSION` to stdout; README documents it and
`--help` (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/15)
- `scan` stops cleanly on `SIGINT` or `SIGTERM`: commits what it has hashed,
deletes nothing more, exits 1 (2026-10-04,
https://git.eeqj.de/sneak/sfdupes/issues/5)
- `report` and `trees` stream the records instead of holding them all in memory;
the schema gains the `files_signature` index (2026-10-04,
https://git.eeqj.de/sneak/sfdupes/issues/14)
- progress prints at once on a non-terminal, uses a real terminal test, and
prints warnings through a spinner instead of racing its redraw (2026-10-03,
https://git.eeqj.de/sneak/sfdupes/issues/13)
- warn about and skip symlink, socket, FIFO, device and `.zfs` operands, keeping
the records beneath them (2026-10-03,
https://git.eeqj.de/sneak/sfdupes/issues/9) https://git.eeqj.de/sneak/sfdupes/issues/9)
- `scan` holds a lock on a lock file beside the database for its whole run, so a - `scan` holds a lock on a lock file beside the database for its whole run,
second `scan` fails at once with exit 1 (2026-10-03, so a second `scan` fails at once with exit 1 (2026-10-03,
https://git.eeqj.de/sneak/sfdupes/issues/53) https://git.eeqj.de/sneak/sfdupes/issues/53)
- test stdout write failures in `report` and `trees`; README states that - test stdout write failures in `report` and `trees`; README states that
`| head` ends sfdupes by `SIGPIPE` and `>&-` writes to `/dev/null` `| head` ends sfdupes by `SIGPIPE` and `>&-` writes to `/dev/null`
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/30) (2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/30)
- `report` and `trees` open the database read-only, and `scan` leaves it out of - `report` and `trees` open the database read-only, and `scan` leaves it
WAL mode, so reading needs only read access (2026-10-03, closes out of WAL mode, so reading needs only read access (2026-10-03, closes
https://git.eeqj.de/sneak/sfdupes/issues/8) https://git.eeqj.de/sneak/sfdupes/issues/8)
- escape tabs, newlines, carriage returns and backslashes in report, trees and - escape tabs, newlines, carriage returns and backslashes in report,
warning paths; the root directory's path is `/` (2026-10-03, trees and warning paths; the root directory's path is `/`
https://git.eeqj.de/sneak/sfdupes/issues/7) (2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/7)
- stamp the git tag or short commit in a plain `docker build .` instead of `dev` - stamp the git tag or short commit in a plain `docker build .`
(2026-10-02, branch `next`, closes instead of `dev` (2026-10-02, branch `next`, closes
https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` sends `.git` https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now
without `.git/config`; the build stage stamps the `VERSION` build argument, sends `.git`, without `.git/config`, and the `Dockerfile` build
else `git describe --tags --always`, and fails if the context carries `.git` stage takes the `VERSION` build argument when one is given,
and the version is still empty, `dev` or `unknown`. CI checks out the full otherwise `git describe --tags --always` of that `.git`. The build
history (`fetch-depth: 0`) so it stamps the same value as `make build`. fails if the context carries `.git` and the version still comes out
empty, `dev` or `unknown`. The CI checkout step fetches the full
history (`fetch-depth: 0`) so CI sees the tag and stamps the same
value as `make build`.
- replace the 1 KiB end-window sampling with the head/tail plus content-hash - replace the 1 KiB end-window sampling with the head/tail plus
ladder (2026-09-22, branch `next`, closes content-hash ladder (2026-09-22, branch `next`, closes
https://git.eeqj.de/sneak/sfdupes/issues/61); README "Duplicate detection" https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is
documents every rung. A file under 10 MiB is hashed in full, and its `head`, hashed in full and compared directly, with no end-window step — its
`tail` and `content` all hold that hash. A larger file gets only its 64 KiB `head`, `tail`, and `content` all hold the whole-file hash. A file at
`head` and `tail` in the hash phase; the content phase, after the update 10 MiB or above gets only the 64 KiB `head` and `tail` in the hash
phase, reads it for `content` (the whole file below 50 MiB, gigabyte-spaced 1 phase; a new content phase, after the update phase, reads it for its
MiB samples at or above) only when its size, `head` and `tail` match another `content` hash — the whole file below 50 MiB, gigabyte-spaced 1 MiB
record's from this scan or an earlier one, and never reads a file gone or samples at or above — only when its size, `head`, and `tail` match
changed since its record was written. `report` and `trees` leave out any another record's, from the same scan or stored by an earlier one, so
record without a `content` hash. The `content` column is part of the version 1 a stored file gains its content hash when it gains a match. A file
schema. that is gone or has changed since its record was written is not
read. The `content` column is part of the version 1 schema. `report`
and `trees` group by the extended signature and leave out any record
without a `content` hash, so the ladder is applied across the whole
database. README "Duplicate detection" documents every rung including
the probabilistic large-file path.
- remove the dead `files.dat` references from `Makefile`, `.gitignore` and - remove the dead `files.dat` references from `Makefile`, `.gitignore`
`.dockerignore` (2026-09-21, branch `next`, closes and `.dockerignore` (2026-09-21, branch `next`, closes
https://git.eeqj.de/sneak/sfdupes/issues/22) https://git.eeqj.de/sneak/sfdupes/issues/22)
- fix the lint-image pin comments and `FROM` form in `Dockerfile` and - fix the lint-image pin comments and `FROM` form in `Dockerfile` and
`Dockerfile.lint` (2026-08-10, branch `next`, closes `Dockerfile.lint` (2026-08-10, branch `next`, closes
https://git.eeqj.de/sneak/sfdupes/issues/25): both pins are now the policy https://git.eeqj.de/sneak/sfdupes/issues/25): dropped the false
`# image:vX.Y.Z, YYYY-MM-DD` comment over a bare `FROM image@sha256:...`, `(Debian-based)` parenthetical (v2.12.1 was Debian too) and the
without the false `(Debian-based)` note or the tag; digest unchanged. redundant tag, so both pins are the policy `# image:vX.Y.Z,
`script/verify-lint-image-pin` still matches the tagless form, and a tag on YYYY-MM-DD` comment over a bare `FROM image@sha256:...`. Digest
one side only is caught as a plain mismatch. unchanged. `script/verify-lint-image-pin` parses those `FROM` lines
and still matches the tagless form; its advice line lost the now
meaningless "tag and digest". With no tag in either reference, a
tag-only disagreement no longer exists — a one-sided tag is caught as
a plain mismatch.
- run all linting in Docker via `Dockerfile.lint` and `script/lint` (2026-08-10, - run all linting in Docker via `Dockerfile.lint` and `script/lint`
branch `next`, closes https://git.eeqj.de/sneak/sfdupes/issues/46): per the (2026-08-10, branch `next`, closes
owner ruling the linter is never installed on a host. `Dockerfile.lint` copies https://git.eeqj.de/sneak/sfdupes/issues/46): per the owner ruling, the
the repo into the digest-pinned `golangci/golangci-lint:v2.12.2` image and linter runs inside a container invoked through the `script/`
runs `golangci-lint config verify` and `golangci-lint run` as build steps; entrypoint and is never installed on a host. New root
`script/lint` builds it. `script/bootstrap` no longer installs or pins the `Dockerfile.lint` COPYs the repo into the digest-pinned
linter, and warns rather than fails when `docker` is absent; `golangci/golangci-lint:v2.12.2` image and runs
`ENV PATH=/home/builder/go/bin:$PATH` went with its `go install`. `golangci-lint config verify` and `golangci-lint run` as build
`script/verify-linter-pin` is retired; `script/verify-lint-image-pin`, a gate steps, so a successful build IS a clean lint; `script/lint` is
in both files, compares their two `FROM` lines and restates neither pin. reduced to building it. `script/bootstrap` loses the `go install`,
Traps: an unchanged tree lets a lint build pass in under a second having run the pin constants, the version parser and `verify_golangci_lint`
no linter, so every gate `RUN` references `ARG CHECK_EPOCH` (BuildKit hashes outright rather than hardening them — with nothing linting on the
the expanded command) and `script/lint` passes `"$(date +%s)-$$"`, the PID host, the `$GOPATH/bin` versus `PATH` problem that motivated them has
because two runs land in the same second easily. Nothing inside an image build no subject — and now warns rather than fails when `docker` is absent.
may shell out to docker, so the `Dockerfile` lint stage calls `golangci-lint` Two traps handled. A lint build on an unchanged tree returns success
directly and the build stage runs `make test` and `make fmt-check` instead of in well under a second having run no linter, which is
`make check`, through `make` because the Makefile's `export CGO_ENABLED = 0` https://git.eeqj.de/sneak/sfdupes/issues/32 and
only reaches what it invokes. `COPY --from=lint /src/go.sum /dev/null` https://git.eeqj.de/sneak/sfdupes/issues/39 again, so
replaces the copied linter binary as the only edge making the build stage wait `Dockerfile.lint` carries `ARG CHECK_EPOCH` referenced
for lint; dropping it would end fail-fast linting under a still-green build. inside every gate `RUN` (BuildKit hashes the expanded command, not
`golangci-lint config verify`, included per the ruling, validates from an the declaration) and `script/lint` passes `"$(date +%s)-$$"` — the
embedded schema with no network call, but `go mod download` above the gates PID matters because two lint runs land inside the same second easily.
still needs the network on a cold cache. Verified: `make lint` green with no And nothing inside an image build may shell out to docker, so the
`golangci-lint` on `PATH`; two back-to-back `script/lint` runs on an untouched main `Dockerfile`'s lint stage now invokes `golangci-lint` directly
tree both ran the linter (27.7s and 28.7s in the lint step, `COPY . .` instead of `make lint`, and its build stage runs `make test` and
`CACHED` above); a planted unused variable failed `script/lint`, and failed `make fmt-check` instead of the `make check` aggregate (`make`, not
`make docker` at `[lint 9/9]` with the build stage stopped at the scripts bare, because the Makefile's `export CGO_ENABLED = 0`
`[builder 3/12]`; the drift guard fails on a tag-only, a digest-only and an only reaches what it invokes). `COPY --from=lint`
unreadable reference, naming both sides; under `--network none` config verify `/usr/bin/golangci-lint` is replaced by
passes a valid config and rejects an invalid one; `make docker` green in 5m35s `COPY --from=lint /src/go.sum /dev/null`: the copied binary was the
with all six gates run under one epoch (lint 37.6s, test 25.2s reporting only edge forcing BuildKit to finish linting before the build stage
`ok sneak.berlin/go/sfdupes 1.938s coverage: 88.5%`, not `(cached)`); in the starts, and dropping it without replacing the edge would have ended
builder image with the Go test cache off, `--user 0:0` still fails fail-fast linting silently under a still-green build. That is
`TestScanHardlinkRunFailsTogether` where the unprivileged user passes. Noted canonical `REPO_POLICIES.md:107`'s ordering edge, restored.
for follow-up, not fixed here: `golangci-lint` warns that `gomodguard` is `ENV PATH=/home/builder/go/bin:$PATH` is gone with the `go install`
deprecated since v2.12.0 in favour of `gomodguard_v2`. that justified it. `script/verify-linter-pin` is retired, deleted
along with its README entry, because both of its subjects ceased to
exist in the same change: it compared a linter binary against
`GOLANGCI_LINT_VERSION` in `script/bootstrap`, and there is now
neither a binary crossing between stages nor a version pin in
bootstrap. The drift it guarded has not gone away, it has moved — the
linter is still pinned twice, now as the `FROM` line of
`Dockerfile.lint` and the `FROM` line of the `Dockerfile` lint stage,
with nothing syncing them, which is exactly what
https://git.eeqj.de/sneak/sfdupes/issues/42 made a build failure. Its
replacement is one new `script/verify-lint-image-pin`,
run as a gate in both files, which compares the two references to
each other and deliberately restates neither: a hardcoded expected
digest would be a third copy and the same drift one file further out.
`golangci-lint config verify` is included per the ruling, and the
concern about its unpinned live HTTPS schema fetch was measured
rather than assumed — under `--network none` the pinned binary both
passes a valid config and rejects an invalid one with the jsonschema
error, so it validates from an embedded schema and makes no network
call of its own. The README scopes that to the gate steps rather
than to linting as a whole: `Dockerfile.lint` runs `go mod download`
above them, so a cold cache still needs the network and only a warm
one lints offline. Verified: `make lint` green with every `PATH`
directory containing a `golangci-lint` removed
(`/home/user/go/bin`, `/home/user/.local/bin`, `/usr/local/bin`;
`command -v golangci-lint` empty); two consecutive `script/lint` runs
on an untouched tree both executed the linter, 27.7s and 28.7s in the
lint step under distinct epochs with the `COPY . .` layer `CACHED`
above them, at 42.2s and 41.8s wall clock — the no-cache rule was not
weakened to shorten that. Negative control: a planted
`var unusedIssue46Sentinel = 1` failed `script/lint` with
`report.go:173:5: var unusedIssue46Sentinel is unused (unused)`, and
failed `make docker` at `[lint 9/9]` with the build stage stopped at
`[builder 3/12]` — `COPY --from=lint`, `script/bootstrap`, the test
gate and `make build` all zero occurrences — then reverted clean. The
drift guard fails on a tag-only disagreement, on a digest-only
disagreement, and on an unreadable reference, naming both sides.
`make docker` green in 5m35s with all six gates executing under one
epoch (lint 37.6s, test 25.2s reporting
`ok sneak.berlin/go/sfdupes 1.938s coverage: 88.5%`, not `(cached)`).
The non-root quirk still holds: in the builder image with the Go test
cache off, `--user 0:0` fails `TestScanHardlinkRunFailsTogether`
(exit 1) where the unprivileged user passes (exit 0). Noted for
follow-up, not fixed here: `golangci-lint` warns that the
`gomodguard` linter is deprecated since v2.12.0 in favour of
`gomodguard_v2`.
- install the Docker build stage's prerequisites by running `script/bootstrap` - install the Docker build stage's prerequisites by running
instead of `apk add --no-cache make` inline (2026-08-09, branch `script/bootstrap` instead of `apk add --no-cache make` inline
`dockerfile-bootstrap`, closes https://git.eeqj.de/sneak/sfdupes/issues/42): (2026-08-09, branch `dockerfile-bootstrap`, closes #42): canonical
the stage copies `script/` plus `go.mod`/`go.sum` and runs `script/bootstrap`, `REPO_POLICIES.md:97` requires it, and the inline install left the
which ends in `go mod download`, so the separate call to it is gone. build stage maintaining its own notion of the toolchain — exactly
`COPY --from=lint /usr/bin/golangci-lint` stays and moves above the bootstrap the divergence #24 exists to close, one layer down. The stage now
layer: it is the only edge making this stage depend on the lint stage, so copies `script/` plus `go.mod`/`go.sum` and runs `script/bootstrap`,
deleting it would end fail-fast linting silently. A new which ends in `go mod download`, so the separate invocation of that
`script/verify-linter-pin`, run in the build stage before bootstrap, fails the is gone. `COPY --from=lint /usr/bin/golangci-lint` stays, and moves
build naming both versions unless that copied binary is the version above the bootstrap layer. It is the only edge making this stage
`script/bootstrap` pins; a pin it cannot read is a hard failure, not a skip. depend on the lint stage, so deleting it as redundant would end
`$GOPATH/bin` joins `PATH`, where bootstrap's `go install` lands. Everything fail-fast linting silently. Letting bootstrap install its own linter
added sits above `ARG CHECK_EPOCH`, and the `chown` and `USER builder` still here would have reintroduced the second toolchain and paid for a
precede `make check`. Verified: the guard fails the build with both versions from-source build of it. What makes the two stages provably one
named when the lint stage's linter is faked to another version, and passes an toolchain rather than two that happen to agree is a new
unmodified build; bootstrap runs clean under Alpine's `sh` and `apk`, finding `script/verify-linter-pin`, run in the build stage on the binary
the copied linter already at the pin; a second build served the bootstrap and that arrives from the lint stage, before bootstrap: it fails the
dependency layers `CACHED` while both gates ran with a fresh epoch; a planted build naming both versions unless that binary is the version
`unused` finding failed the build at the lint gate in 48.9s with the build `script/bootstrap` pins. Bootstrap's own check could not serve that
stage's `make check` never starting; and the suite run in the image as purpose — it reinstalls its pin from source and then verifies
`--user 0:0` fails `TestScanHardlinkRunFailsTogether`, so the drop to the whatever `PATH` resolves, so drift self-heals silently and a lint
unprivileged user is still needed. That last check needs the Go test cache stage image bumped on its own would lint at the new version while
off: as root it first reported `ok ... (cached)`, reusing the build-time `make check` ran at the old one, green. The linter version is pinned
result. Build times on a noisy shared host: 2m13s on an unchanged tree, 2m17s in two independent places (the lint stage image digest and
and 4m29s after a source change, 5m14s cold, which breaches the policy `GOLANGCI_LINT_VERSION`) and nothing else keeps them in sync, so a
ceiling; `chown -R builder:builder /src /home/builder` walks the module cache half-applied bump is now a build failure. The pin is read out of
and alone varied from 77s to 210s across those builds, and `main` measured `script/bootstrap`, which stays the single source of truth; a pin
5m03s cold with a 209s `chown`. Filed as that cannot be read is a hard failure, not a skip. The check needs
https://git.eeqj.de/sneak/sfdupes/issues/43 no `CHECK_EPOCH`: its only inputs are the copied binary and
- bust the Docker layer cache for the gate steps, so `script/cibuild` and `script/`, so Docker invalidates the layer exactly when a cached
`script/docker` cannot report a green they did not earn (2026-08-09, branch result would stop being true, and it is documented with the other
`cibuild-cache-bust`, closes https://git.eeqj.de/sneak/sfdupes/issues/32): the entrypoints in the README. `$GOPATH/bin` joins `PATH` because
`Dockerfile` copies the tree before its gates, so on an unchanged tree Docker that is where bootstrap's `go install` lands and bootstrap verifies
served them from cache and the build exited 0 having run nothing. Both scripts its installs against what `PATH` resolves — nothing in the image is
now pass `--build-arg CHECK_EPOCH="$(date +%s)"`. `ARG` is per stage and the shadowed by it, the directory does not exist until bootstrap runs.
gates span two stages, so it is declared in both; BuildKit hashes the expanded Everything added sits above `ARG CHECK_EPOCH`, and the `chown` and
command, so each gate `RUN` echoes the epoch, which also logs it as evidence `USER builder` still precede `make check`. Verified: the guard fails
the layer ran. It sits below the dependency layers so they stay cached. the build with both versions named when the lint stage's linter is
Verified under `BUILDKIT_PROGRESS=plain`, each script run twice back to back faked to a different version, and an unmodified build still passes
on an unchanged tree: all three gates ran on all four runs with a fresh epoch it; bootstrap runs clean under Alpine's `sh` and its `apk` branch,
(`script/cibuild` 78.8s then 61.1s; `script/docker` 61.1s then 53.4s), and installing `git` and `make` and finding the copied
thirteen steps were still served `CACHED`. With a planted `unused` finding the linter already at the pin; a second build served the bootstrap and
build failed at `make lint` in 36.1s and the build-stage `make check` never dependency layers `CACHED` while both gates ran with a fresh epoch;
started. Run as root, the same image fails `TestScanHardlinkRunFailsTogether`, a planted `unused` finding failed the build at the lint gate in
because root reads through the `chmod(0)` the test relies on, so the build 48.9s with the build stage's `make check` never starting; and the
stage must drop to the unprivileged `builder` user. Local fix only; suite run in the image as `--user 0:0` fails
propagating it to the canonical templates is `TestScanHardlinkRunFailsTogether`, so the drop to the unprivileged
https://git.eeqj.de/sneak/prompts/issues/26 user is still load-bearing. That last check needs the Go test cache
- check the installed golangci-lint version in `script/bootstrap` instead of disabled — the first attempt reported `ok ... (cached)` as root,
only its presence (2026-08-09, branch `bootstrap-version-check`, closes reusing the result the build-time run had left in the shared cache,
https://git.eeqj.de/sneak/sfdupes/issues/24): the version lives only in which would have read as a pass. Build wall time, on a shared host
`GOLANGCI_LINT_VERSION`, with the `go install` module ref derived from it, and running many concurrent builds and so noisy: 2m13s on an unchanged
any installed version that is not the pin — older, newer, absent or tree, 2m17s and 4m29s for two builds after a source change, 5m14s
unparseable — is reinstalled. `go install` writes into `GOBIN` (or cold. Only the cold one breaches the policy ceiling, and not because
`GOPATH/bin`) while `make lint` runs the first `golangci-lint` on `PATH`, so of this change — `chown -R builder:builder /src /home/builder` walks
bootstrap re-reads the effective version after installing and, on a mismatch, the module cache and re-runs on every source change, and it alone
prints both paths and both versions and exits non-zero; it does not reorder varied between 77s and 210s across those four builds, which is also
`PATH` or delete anyone's binary. The `--version` call keeps its stderr and is the whole spread in the totals. The same cold measurement against
bounded by `timeout(1)` where that exists. `git`, `make` and `go` keep `main` is 5m03s with a 209s `chown`. Filed as #43
presence-only checks. Verified by bootstrapping this host from v2.10.1 to - bust the Docker layer cache for the gate steps, so `script/cibuild`
v2.12.2 and again to a no-op, and by stub runs of the script under `dash` and `script/docker` cannot report a green they did not earn
covering a thirteen-input version-parse matrix, a shadowed install that must (2026-08-09, branch `cibuild-cache-bust`, closes #32): both scripts
exit non-zero, an install destination not on `PATH`, `GOBIN` set, and a wedged were bare `docker build` invocations with no cache control, and the
binary that must hit the timeout; `make check` and `make lint` are clean at `Dockerfile` copies the tree before running its gates, so on an
v2.12.2, so v2.10.1 was not hiding any findings on `main` unchanged tree Docker served those layers from cache and the build
exited 0 having executed nothing. That is not hypothetical here —
every merge this repo has done is a non-fast-forward merge of an
undiverged branch, so each merge commit's tree is byte-identical to
the branch head's and each merge CI run was almost certainly a full
cache hit; and PR #31's reviewer found `make docker` returning
success as a 17-layer cache hit, catching it only by being
suspicious. The fix is `ARG CHECK_EPOCH` with the scripts passing
`--build-arg CHECK_EPOCH="$(date +%s)"`. Two details make or break
it. `ARG` is scoped per stage and this `Dockerfile` has three gates
across two — `make fmt-check` and `make lint` in the lint stage,
`make check` in the build stage — so a single declaration would have
left one stage silently cacheable; it is declared in both. And
BuildKit hashes the expanded command, not the declaration, so a
declared-but-unreferenced `ARG` invalidates nothing: each gate `RUN`
echoes the epoch, which also puts the value in the build log as
evidence the layer really ran. Placement is below the dependency
layers on purpose — a build that goes cold every time would be a
different bug, not a fix. Verified by running each script twice back
to back on an unchanged tree under `BUILDKIT_PROGRESS=plain`: all
three gates executed on all four runs, each with a fresh epoch in
the log (`script/cibuild` 78.8s then 61.1s; `script/docker` 61.1s
then 53.4s), and twelve steps were still served `CACHED` in the
steady state — both `go mod download`s, `apk add`, `adduser`, the
`chown`, every `go.mod`/`go.sum` and source copy, the linter copy
out of the lint stage, and the binary copy into the runtime stage.
The lint stage still gates the build stage: with a deliberate
`unused` finding planted in the tree, the build failed at
`make lint` in 36.1s and the build-stage `make check` never started.
The build stage also still drops to the unprivileged `builder` user
before `make check`, which the suite depends on rather than merely
prefers: forcing the same image to run the tests as root fails
`TestScanHardlinkRunFailsTogether`, because root reads straight
through the `chmod(0)` the test uses to prove hard links are read
once. This is the local fix only; propagating it to the canonical
templates is `prompts` #26
- check the installed golangci-lint version in `script/bootstrap`
instead of only its presence (2026-08-09, branch
`bootstrap-version-check`, closes #24): `missing golangci-lint` meant
any linter already on `PATH` satisfied the check, so the pin was never
consulted and the v2.12.2 bump from #3 was inert on every host that
already had one — this host ran v2.10.1 against a v2.12.2 pin,
`make check` went green, and `make docker` then rejected the same
commit with findings the local gate never saw. The version now lives
in one place, `GOLANGCI_LINT_VERSION`, with the `go install` module
ref derived from it so a bump cannot half-apply; a
`golangci_lint_version` helper parses `golangci-lint --version`
(taking the field after the word `version` and tolerating an optional
leading `v`, which the module ref carries and the binary's output does
not), and any version that is not the pin — older, newer, absent or
unparseable — is reinstalled. The install is then verified against the
binary `PATH` actually resolves: `go install` writes into `GOBIN` (or
`GOPATH/bin`) while `make lint` runs whichever `golangci-lint` comes
first on `PATH`, so a wrong-version one sitting ahead of it — nix,
apt, brew, apk, or the `/usr/local/bin` copy the `Dockerfile` builder
stage makes — would swallow the install and leave the local gate
disagreeing with CI under an affirmative `bootstrap complete`.
Bootstrap now re-reads the effective version after installing and, on
a mismatch, prints both paths and both versions to stderr and exits
non-zero instead of claiming success; it does not reorder anyone's
`PATH` or delete their binary. The `--version` call keeps its stderr
connected, so a present-but-broken binary says why rather than
reinstalling forever in silence, and is bounded by `timeout(1)` where
that exists, so a wedged binary cannot hang bootstrap. `git`, `make`
and `go` keep their presence-only checks and now say why in a
comment: they are host package-manager tools the repo deliberately
does not pin, with `go.mod` governing the language version and the
digest-pinned images covering reproducible builds. Verified on this
host by bootstrapping from v2.10.1 to v2.12.2 and running it again to
a no-op, plus stub runs of the real script under `dash` covering a
thirteen-input parse matrix (absent, older, newer, host-style,
image-style, leading-`v`, stderr-only, empty, non-zero exit, impostor
binary, `(devel)`, trailing `version`), a shadowed install that must
exit non-zero, an install destination not on `PATH` at all, `GOBIN`
set, and a wedged binary that must hit the timeout; `make check` and
`make lint` are clean at v2.12.2, so v2.10.1 was not hiding any
findings on `main`
- unwind the hash worker pool on the error path (2026-08-09, branch - unwind the hash worker pool on the error path (2026-08-09, branch
`hash-pool-cleanup`, closes https://git.eeqj.de/sneak/sfdupes/issues/6): the `hash-pool-cleanup`, closes #6): `hashPhase` used to return the
pool is now an owned, context-aware `hashPool`: every blocking send in the moment `recordRun` failed and abandon the pool — the feeder parked
feeder and the workers selects on `ctx.Done()`, `jobs` is closed on every path forever on a full `jobs` channel and every worker on a full
out, and `hashPhase` defers `pool.stop()`, which cancels and then drains `results` channel. That only stopped being invisible when #4 landed
`results` until the last goroutine has exited — draining is what frees a and `runScan` began unwinding instead of calling `os.Exit`. The
worker already parked on a send. `ctx` is threaded from `cmd.Context()` pool is now an owned, context-aware `hashPool`: every blocking send
through `runScan`, `syncScan`, both worker pools and the whole database layer, in the feeder and the workers selects on `ctx.Done()`, `jobs` is
as the first parameter everywhere. The walk pool gets the same treatment plus closed on every path out, and `hashPhase` defers `pool.stop()`,
a `ctx.Err()` guard after the walk: a cancelled walk yields a partial size which cancels and then drains `results` until the last goroutine
census, and every file it never reached looks vanished to the update phase. has exited — draining is what frees a worker already parked on a
That phase's own `BeginTx` also fails on the cancelled context before deleting send. `ctx` is threaded from `cmd.Context()` through `runScan`,
anything, but the guard is the barrier that still holds once an interrupted `syncScan`, both worker pools and the whole database layer (it is
scan may commit what it has. Tests drive `run(scan)` against a database whose the first parameter everywhere), so #5 can hand this path a signal
insert trigger aborts and assert that the scan fails instead of hanging and and needs to add nothing else. The walk pool never leaked, because
that `runtime.NumGoroutine()` polls back to its pre-scan baseline; others `walkPhase` always drains its events to close, but it has the same
cancel a scan part-way through the walk, deterministically, by counting its unbounded-send shape and #5 will give it an early return, so it
own consultations of `ctx.Done()`, and assert that it stops at the guard gets the same treatment plus a `ctx.Err()` guard after the walk: a
holding a partial census and a still-populated record index, with every record cancelled walk yields a partial size census, and every file it never
intact. Direct tests of `sendEvent`, the walk workers, `dispatchDirs`, reached looks vanished to the update phase. That phase's own
`feedHashJobs`, `hashWorker` and `hashPhase` cover the remaining cancellation `BeginTx` fails on the same cancelled context before deleting
branches of both pools anything, so the guard is defence in depth rather than the only
- guarantee the database is closed on every fatal exit path (2026-08-09, branch barrier — but it is the one that survives #5 deciding an interrupted
`db-close-on-fatal`, closes https://git.eeqj.de/sneak/sfdupes/issues/4): scan may commit what it has. Tests drive `run(scan)` against a
`fatalf` and its `os.Exit(1)` are gone, so the deferred `db.Close()` — and database whose insert trigger aborts, and assert both that the scan
with it the SQLite WAL checkpoint — now actually runs when a subcommand fails; fails instead of hanging and that `runtime.NumGoroutine()` polls
`runScan`, `runReport`, `runTrees`, `loadRecords` and `resolveRoots` return back to its pre-scan baseline; a second set cancels a scan part-way
errors instead. The single exit point is `run` in `main.go`: it maps a through the walk — deterministically, by counting the scan's own
`fatalError` (anything a subcommand returned) to exit 1 and cobra's own consultations of `ctx.Done()` rather than racing a timer — and
argument and flag errors to exit 2, which keeps a runtime failure from being asserts that it stops at the guard holding a partial census and a
reported as a usage error or printing the usage text. New `main_test.go` still-populated record index, with every record intact. The
drives the CLI in-process and asserts the exit codes from README §Error remaining cancellation branches of both pools are covered by direct
handling plus the stdout/stderr split, including that a fatal error raised tests of `sendEvent`, the walk workers, `dispatchDirs`,
after the database is open leaves no `-wal`/`-shm` sidecar behind for `scan`, `feedHashJobs`, `hashWorker` and `hashPhase`
`report` or `trees` - guarantee the database is closed on every fatal exit path
- update golangci-lint to v2.12.2 with the canonical config (2026-08-09, branch (2026-08-09, branch `db-close-on-fatal`, closes #4): `fatalf` and
`golangci-v2.12.2`, merged as `38a01bd`, closes its `os.Exit(1)` are gone, so the deferred `db.Close()` — and with
https://git.eeqj.de/sneak/sfdupes/issues/3): bumped the pinned linter in the it the SQLite WAL checkpoint — now actually runs when a subcommand
`Dockerfile` lint stage and `script/bootstrap` from v2.12.1 to v2.12.2, and fails; `runScan`, `runReport`, `runTrees`, `loadRecords` and
replaced `.golangci.yml` with the canonical file — the linter settings (`lll`, `resolveRoots` return errors instead. The single exit point is `run`
`funlen`, `cyclop`, `dupl` thresholds) now live under `linters.settings` per in `main.go`: it maps a `fatalError` (anything a subcommand
the v2 schema, so they are actually applied; no new lint findings surfaced returned) to exit 1 and cobra's own argument and flag errors to exit
- convert Makefile targets to scripts-to-rule-them-all `script/` entrypoints 2, which keeps a runtime failure from being reported as a usage
like the other managed repos (2026-07-26, commit `3abeacf`, closes error or printing the usage text. New `main_test.go` drives the CLI
https://git.eeqj.de/sneak/sfdupes/issues/1): all 12 `script/` entrypoints in-process and asserts the exit codes from README §Error handling
exist (`bootstrap`, `setup`, `projectname`, `test`, `lint`, `fmt`, plus the stdout/stderr split, including that a fatal error raised
`fmt-check`, `check`, `docker`, `cibuild`, `precommit`, `install-precommit`) after the database is open leaves no `-wal`/`-shm` sidecar behind
and every Makefile target is now a thin shim over them for `scan`, `report` or `trees`
- update golangci-lint to v2.12.2 with the canonical config
(2026-08-09, branch `golangci-v2.12.2`, merged as `38a01bd`,
closes #3): bumped the pinned linter in the `Dockerfile` lint
stage and `script/bootstrap` from v2.12.1 to v2.12.2, and replaced
`.golangci.yml` with the canonical file — the linter settings
(`lll`, `funlen`, `cyclop`, `dupl` thresholds) now live under
`linters.settings` per the v2 schema, so they are actually
applied; no new lint findings surfaced
- convert Makefile targets to scripts-to-rule-them-all `script/`
entrypoints like the other managed repos (2026-07-26, commit
`3abeacf`, closes #1): all 12 `script/` entrypoints exist
(`bootstrap`, `setup`, `projectname`, `test`, `lint`, `fmt`,
`fmt-check`, `check`, `docker`, `cibuild`, `precommit`,
`install-precommit`) and every Makefile target is now a thin shim
over them, matching the other managed repos
- make the binary the default Make target (2026-07-24, branch - make the binary the default Make target (2026-07-24, branch
`make-default-target`): plain `make` now builds `sfdupes` (previously it ran `make-default-target`): plain `make` now builds `sfdupes`
`check` plus `build`); `make build` remains as an alias (previously it ran `check` plus `build`); `make build` remains as
- scan-wide phases, concurrent operands, batched updates (2026-07-24, branch an alias
`scan-wide-phases`): all operands seed the shared walk pool and every pass - scan-wide phases, concurrent operands, batched updates (2026-07-24,
runs once over the whole scan, so totals and ETAs are scan-global; the branch `scan-wide-phases`): all operands seed the shared walk pool
per-operand walk/hash/update cycles and their stderr announcements are gone; and every pass runs once over the whole scan, so totals and ETAs
the update pass commits in batched transactions — the filesystem is are scan-global; the per-operand walk/hash/update cycles and their
authoritative and the database an eventually-consistent reflection, so stderr announcements are gone; the update pass commits in batched
scan-level atomicity is not required transactions — the filesystem is authoritative and the database an
eventually-consistent reflection, so scan-level atomicity is not
required
- split the stat pass back out of the walk (2026-07-24, branch - split the stat pass back out of the walk (2026-07-24, branch
`parallel-phases`): phases are strictly sequential again — walk, stat, hash, `parallel-phases`): phases are strictly sequential again — walk,
update per operand — with parallelism only inside each phase; the walk stat, hash, update per operand — with parallelism only inside each
enumerates paths with per-directory workers and the stat pass lstats them with phase; the walk enumerates paths with per-directory workers and the
per-file workers, restoring the exact total/ETA stat bar stat pass lstats them with per-file workers, restoring the exact
- announce each operand on stderr before its passes (2026-07-24, branch total/ETA stat bar
`scan-operand-progress`): with per-operand walk/hash/update cycles, a - announce each operand on stderr before its passes (2026-07-24,
multi-operand run (e.g. `scan /srv/*`) showed pass totals that looked like the branch `scan-operand-progress`): with per-operand walk/hash/update
whole run's cycles, a multi-operand run (e.g. `scan /srv/*`) showed pass totals
- parallel walk (2026-07-24, branch `parallel-walk`): the walk pass was a single that looked like the whole run's — an operator watching operand 3 of
goroutine and took hours at ~20M files on a busy pool (observed: 22M files in 14 hash 300k files concluded 20M files were being skipped
4h on a ZFS server); it is now a per-directory worker-pool traversal that - parallel walk (2026-07-24, branch `parallel-walk`): the walk pass
records size/mtime during the walk (folding away the separate stat pass, was a single goroutine and took hours at ~20M files on a busy pool
halving metadata I/O), and each `PATH` operand commits in its own transaction (observed: 22M files in 4h on a ZFS server); it is now a
so an interrupted scan keeps completed operands per-directory worker-pool traversal that records size/mtime during
the walk (folding away the separate stat pass, halving metadata
I/O), and each `PATH` operand commits in its own transaction so an
interrupted scan keeps completed operands
- persistent scan database (2026-07-24, branch `persistent-database`): `scan` - persistent scan database (2026-07-24, branch `persistent-database`):
now maintains a SQLite database (`modernc.org/sqlite`, pure Go, cgo stays `scan` now maintains a SQLite database (`modernc.org/sqlite`, pure
disabled) keyed by absolute path that survives between runs — a rescan hashes Go, cgo stays disabled) keyed by absolute path that survives between
only new or changed files (by mtime/size), deletes records for files vanished runs — a rescan hashes only new or changed files (by mtime/size),
from under the scanned operands, and leaves records outside them untouched, so deletes records for files vanished from under the scanned operands,
`scan` can be cronned daily; `report` and `trees` read the database (no and leaves records outside them untouched, so `scan` can be cronned
positional arguments) instead of a scan stream. Database at daily; `report` and `trees` read the database (no positional
`/var/lib/sfdupes/db.sqlite`, overridable via `SFDUPES_DATABASE`; WAL arguments) instead of a scan stream. Database at
journaling plus a single-transaction update keep a report run during a scan `/var/lib/sfdupes/db.sqlite`, overridable via `SFDUPES_DATABASE`;
safe WAL journaling plus a single-transaction update keep a report run
- add the `origin` remote (`git@git.eeqj.de:sneak/sfdupes.git`), tag `v0.0.1`, during a scan safe
and push `main` plus tags (2026-07-23) - add the `origin` remote (`git@git.eeqj.de:sneak/sfdupes.git`), tag
`v0.0.1`, and push `main` plus tags (2026-07-23)
- `scan` CLI rework (2026-07-23, branch `scan-required-paths`): required - `scan` CLI rework (2026-07-23, branch `scan-required-paths`): required
`PATH...` operands via cobra flags replacing the `/srv` `-root` default; new `PATH...` operands via cobra flags replacing the `/srv` `-root`
`-x`/`--one-file-system` flag (GNU convention) to stop at filesystem default; new `-x`/`--one-file-system` flag (GNU convention) to stop
boundaries, which are crossed by default at filesystem boundaries, which are crossed by default
- bring the repo into full policy compliance (2026-07-23, branch - bring the repo into full policy compliance (2026-07-23, branch
`repo-policy-compliance`; checklist below) `repo-policy-compliance`; checklist below)
- `git init` with README-only first commit; code baseline committed on `main` - `git init` with README-only first commit; code baseline committed on
(2026-07-22) `main` (2026-07-22)
- implement `scan`, `report`, and `trees` subcommands (pre-git history) - implement `scan`, `report`, and `trees` subcommands (pre-git history)
# Future Steps # Future Steps
- possible later features (explicitly out of scope per README): full-content - possible later features (explicitly out of scope per README):
verification of candidates, removal-script helpers full-content verification of candidates, removal-script helpers
# Repo Policy Compliance # Repo Policy Compliance
Audited 2026-07-22 against `REPO_POLICIES.md` (2026-07-06), the existing repo Audited 2026-07-22 against `REPO_POLICIES.md` (2026-07-06), the existing
checklist, and the Go styleguide. Code is already gofmt-clean, so no standalone repo checklist, and the Go styleguide. Code is already gofmt-clean, so no
formatting commit is needed. standalone formatting commit is needed.
- [x] `.gitignore` missing — the compiled `sfdupes` binary and `files.dat` sit - [x] `.gitignore` missing — the compiled `sfdupes` binary and
untracked in the tree; needs OS/editor/Go artifacts plus secrets patterns `files.dat` sit untracked in the tree; needs OS/editor/Go
artifacts plus secrets patterns
- [x] `.editorconfig` missing - [x] `.editorconfig` missing
- [x] `LICENSE` missing and README has no License section (MIT assumed from - [x] `LICENSE` missing and README has no License section (MIT assumed
house convention — user to confirm) from house convention — user to confirm)
- [x] `REPO_POLICIES.md` missing from repo root - [x] `REPO_POLICIES.md` missing from repo root
- [x] `.golangci.yml` missing (install canonical copy); code must then pass - [x] `.golangci.yml` missing (install canonical copy); code must then
`make lint` (150 findings fixed; `make lint` is clean) pass `make lint` (150 findings fixed; `make lint` is clean)
- [x] `Makefile` lacks required targets `test`, `lint`, `fmt`, `fmt-check`, - [x] `Makefile` lacks required targets `test`, `lint`, `fmt`,
`docker`, `hooks`; `check` currently depends on `build`, which writes the `fmt-check`, `docker`, `hooks`; `check` currently depends on
binary (`make check` must not modify files) `build`, which writes the binary (`make check` must not modify
- [x] no tests — `go test ./...` has nothing to run; policy requires real tests files)
with a 30-second timeout and the conditional `-v` rerun pattern (suite - [x] no tests — `go test ./...` has nothing to run; policy requires
covers parsing, grouping, digests, suppression, hashing, and the scan real tests with a 30-second timeout and the conditional `-v`
pipeline; 64% coverage) rerun pattern (suite covers parsing, grouping, digests,
- [x] `Dockerfile` missing — Go multistage with hash-pinned images: fail-fast suppression, hashing, and the scan pipeline; 64% coverage)
lint stage, build stage running `make check` - [x] `Dockerfile` missing — Go multistage with hash-pinned images:
fail-fast lint stage, build stage running `make check`
- [x] `.dockerignore` missing - [x] `.dockerignore` missing
- [x] `.gitea/workflows/check.yml` missing (`docker build .` on push, checkout - [x] `.gitea/workflows/check.yml` missing (`docker build .` on push,
action pinned by commit SHA) checkout action pinned by commit SHA)
- [x] README lacks required sections: Description first line - [x] README lacks required sections: Description first line
(name/purpose/category/license/author), Getting Started, Rationale, TODO, (name/purpose/category/license/author), Getting Started,
License, Author Rationale, TODO, License, Author
- [x] README non-goal "no git repository setup and no CI" is stale now that the - [x] README non-goal "no git repository setup and no CI" is stale now
repo is under git with CI that the repo is under git with CI
- [x] pre-commit hook not installed (`make hooks` once the target exists) - [x] pre-commit hook not installed (`make hooks` once the target
exists)
Accepted divergences (no action): Accepted divergences (no action):
- flat single-package layout with `.go` files in the repo root — fine for a - flat single-package layout with `.go` files in the repo root — fine
small single-binary tool per the Go styleguide; the tracker audit agrees for a small single-binary tool per the Go styleguide; the tracker
- `make test` runs without `-race` — the repo mandates `CGO_ENABLED=0` (pure-Go audit agrees
builds) and the race detector requires cgo, so the detector runs in a separate - `go test` runs without `-race` — the repo mandates `CGO_ENABLED=0`
cgo-enabled container, `make test-race`, which is not part of `make check` (pure-Go builds) and the race detector requires cgo
+49 -393
View File
@@ -4,35 +4,20 @@ import (
"context" "context"
"database/sql" "database/sql"
"errors" "errors"
"fmt"
"os" "os"
"os/signal"
"path/filepath" "path/filepath"
"slices"
"strconv" "strconv"
"strings"
"sync" "sync"
"sync/atomic" "sync/atomic"
"syscall"
"testing" "testing"
"time" "time"
) )
// This file gathers the tests for scan cancellation and worker-pool // poolUnwind bounds how long a goroutine is given to leave a pool
// unwinding. Everything it exercises lives in scan.go, so by the repo's // after its context is cancelled. Only a failing run ever waits this
// convention of one test file per source file it would belong in // long: a pool that ignored its cancellation parks forever, and this
// scan_test.go. It is kept separate on purpose: cancellation behaviour // is what turns that into a failed assertion instead of a suite that
// cuts across both the walk pool and the hash pool as a single concern, // hangs until the test binary's own timeout.
// and scan_test.go is already over 1,600 lines. That is the deliberate
// exception the convention otherwise expects to be stated.
// poolUnwind bounds how long a test waits for a cancellation to take
// effect: for a goroutine to return or a channel to close once its
// context is cancelled, or for a signal to cancel the scan's context.
// Only a failing run waits this long, and the bound is what makes that
// failure an assertion instead of a hang. A call made without it, as
// most of this file's scans are, has no bound: a regression that parks
// it is caught only as the test binary's own timeout.
const poolUnwind = 2 * time.Second const poolUnwind = 2 * time.Second
// walkClock is a context whose cancellation is driven by the scan's // walkClock is a context whose cancellation is driven by the scan's
@@ -43,14 +28,9 @@ const poolUnwind = 2 * time.Second
// //
// The accounting behind the n chosen by each test: every blocking // The accounting behind the n chosen by each test: every blocking
// channel operation in the walk selects on Done, so the walk spends // channel operation in the walk selects on Done, so the walk spends
// one consultation per file event plus a couple per directory. The // one consultation per file event plus a couple per directory, while
// index load that runs ahead of it also consults Done, but a bounded // the index load that runs ahead of it spends a small fixed number
// number of times that does not grow with the record count. The tests // (three) whatever the record count.
// depend on that property, not on the bound's exact value: each test
// sets n from the consultations of the walk, plus those of the hash
// phase when it cancels mid-hash, far from both ends of the phase it
// interrupts, so the cancellation lands inside that phase whatever the
// record count.
type walkClock struct { type walkClock struct {
n int64 n int64
seen atomic.Int64 seen atomic.Int64
@@ -106,14 +86,11 @@ func (c *walkClock) Value(_ any) any {
// directory still queued and only the handful already in flight can // directory still queued and only the handful already in flight can
// emit anything more. // emit anything more.
const ( const (
walkCancelDirs = 100 walkCancelDirs = 100
walkCancelFilesPerDir = 20 walkCancelFilesPerDir = 20
walkCancelFiles = walkCancelDirs * walkCancelFilesPerDir walkCancelFiles = walkCancelDirs * walkCancelFilesPerDir
walkCancelWorkers = 4 walkCancelWorkers = 4
// The most files the walkCancelWorkers directories already in walkCancelInFlightDirs = walkCancelWorkers * walkCancelFilesPerDir
// flight when the scan is cancelled can still emit, at
// walkCancelFilesPerDir each. A file count, not a directory count.
walkCancelInFlightFiles = walkCancelWorkers * walkCancelFilesPerDir
) )
// walkCancelAtDone is the consultation on which the fixture's context // walkCancelAtDone is the consultation on which the fixture's context
@@ -173,13 +150,9 @@ func assertRecordsIntact(t *testing.T, db *sql.DB, before []string) {
// Every one of those records would look vanished to the update phase. // Every one of those records would look vanished to the update phase.
// The guard is what stops the scan there, and this test is what // The guard is what stops the scan there, and this test is what
// notices if it stops doing so: deleting the guard, or making it // notices if it stops doing so: deleting the guard, or making it
// unreachable, makes the scan carry its truncated view into the update // unreachable, makes the scan carry its truncated view into a later
// phase, which counts every record the walk never reached for removal. // phase and fail there instead, with a wrapped error rather than the
// // bare cancellation.
// The syncScan call here is not bounded by poolUnwind: a regression
// that left a worker pool parked would hang it, and that regression is
// caught only by the test binary's own timeout, not by a quick
// assertion.
// //
//nolint:paralleltest // counts goroutines: must not run beside others //nolint:paralleltest // counts goroutines: must not run beside others
func TestSyncScanCancelledMidWalkKeepsRecords(t *testing.T) { func TestSyncScanCancelledMidWalkKeepsRecords(t *testing.T) {
@@ -209,10 +182,10 @@ func TestSyncScanCancelledMidWalkKeepsRecords(t *testing.T) {
// assertWalkGuardAborted checks that the scan stopped at the post-walk // assertWalkGuardAborted checks that the scan stopped at the post-walk
// guard: with a census that is neither empty (the walk really ran) // guard: with a census that is neither empty (the walk really ran)
// nor complete (it really was cut short), and with no record counted // nor complete (it really was cut short), and with the guard's own
// for removal. A removal count means the partial census was carried // bare cancellation as the error. A wrapped error means the partial
// past the guard into the update phase, which is the failure this test // census was carried past the guard into the hash or update phase,
// exists to catch. // which is the failure this test exists to catch.
func assertWalkGuardAborted(t *testing.T, st scanStats, err error) { func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
t.Helper() t.Helper()
@@ -221,6 +194,12 @@ func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
err, context.Canceled) err, context.Canceled)
} }
if errors.Unwrap(err) != nil {
t.Errorf("syncScan reported %q, want the guard's bare "+
"cancellation: a wrapped error means the truncated census "+
"reached a later phase", err)
}
if st.unchanged == 0 { if st.unchanged == 0 {
t.Fatalf("stats = %+v: the census is empty, so the walk never "+ t.Fatalf("stats = %+v: the census is empty, so the walk never "+
"ran and the guard was reached for the wrong reason", st) "ran and the guard was reached for the wrong reason", st)
@@ -232,11 +211,10 @@ func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
} }
// The workers drop every directory still queued once the scan is // The workers drop every directory still queued once the scan is
// cancelled, so only the files in the directories already in flight // cancelled, so only the directories already in flight can add to
// can add to the census after the fact. A census beyond that bound // the census after the fact. A census beyond that bound would mean
// would mean the cancellation was not observed where it should have // the cancellation was not observed where it should have been.
// been. limit := walkCancelAtDone + walkCancelInFlightDirs
limit := walkCancelAtDone + walkCancelInFlightFiles
if st.unchanged > limit { if st.unchanged > limit {
t.Errorf("census covers %d files, want at most %d: the walk kept "+ t.Errorf("census covers %d files, want at most %d: the walk kept "+
"taking directories off the queue after cancellation", "taking directories off the queue after cancellation",
@@ -282,252 +260,6 @@ func TestSyncScanCancelledBeforeLoadIndex(t *testing.T) {
assertRecordsIntact(t, db, before) assertRecordsIntact(t, db, before)
} }
// hashCancelAtDone is the consultation on which the mid-hash test's
// context cancels itself. The walk of buildWalkCancelTree spends about
// one per file and three per directory, and the hash phase then one per
// file hashed, so this lands about half way through the hash phase.
const hashCancelAtDone = walkCancelFiles + 3*walkCancelDirs +
walkCancelFiles/2
// TestSyncScanCancelledMidHashKeepsHashedRecords cancels a first scan
// part-way through its hash phase. The fixture holds fewer files than a
// batch, so every file hashed is still waiting to be committed: the scan
// must commit them all before it returns, and the next scan must hash
// only the rest.
func TestSyncScanCancelledMidHashKeepsHashedRecords(t *testing.T) {
t.Parallel()
dir := buildWalkCancelTree(t)
db := openTestDB(t)
st, err := syncScan(newWalkClock(hashCancelAtDone), db,
[]string{dir}, walkCancelWorkers, false)
if !errors.Is(err, context.Canceled) {
t.Fatalf("syncScan cancelled mid-hash = %v, want %v",
err, context.Canceled)
}
if st.walked != walkCancelFiles || st.added == 0 ||
st.added >= walkCancelFiles {
t.Fatalf("stats = %+v: want the walk complete and the hash phase "+
"cut short", st)
}
if got := len(dbRecords(t, db)); got != st.added {
t.Errorf("%d records after the cancelled scan, want the %d it hashed",
got, st.added)
}
hashed := st.added
st = syncTree(t, db, dir)
if st.added != walkCancelFiles-hashed || st.unchanged != hashed {
t.Errorf("next scan stats = %+v, want %d added %d unchanged",
st, walkCancelFiles-hashed, hashed)
}
}
// storedPaths opens the database at path as report does, which fails
// unless it is a valid database, and returns its records' paths.
func storedPaths(t *testing.T, path string) []string {
t.Helper()
db, err := openReportDatabase(t.Context(), path)
if err != nil {
t.Fatal(err)
}
defer func() { _ = db.Close() }()
return recordPaths(dbRecords(t, db))
}
// TestRunScanInterrupted calls the scan entrypoint with a context that
// is already cancelled, as when a signal arrives at once. It must return
// errInterrupted promptly with its one line on stderr and nothing on
// stdout, leave the database valid and as it was, and leave nothing in
// the way of the next scan, which must bring the database up to date.
func TestRunScanInterrupted(t *testing.T) {
path := testDBPath(t)
t.Setenv(databaseEnv, path)
stdout := captureStdout(t)
stderr := captureStderr(t)
dir := buildSmokeTree(t)
err := runScan(t.Context(), []string{dir}, walkCancelWorkers, false)
if err != nil {
t.Fatal(err)
}
before := storedPaths(t, path)
// A vanished file and a new one: the interrupted scan records
// neither.
gone := filepath.Join(dir, "a", "unique.bin")
err = os.Remove(gone)
if err != nil {
t.Fatal(err)
}
added := writeFile(t, dir, "a/new.bin", pattern(50, 10))
shown := len(stderr())
done := make(chan struct{})
go func() {
defer close(done)
err = runScan(cancelledContext(t), []string{dir}, walkCancelWorkers,
false)
}()
awaitReturn(t, done, "runScan")
if !errors.Is(err, errInterrupted) {
t.Fatalf("runScan on a cancelled context = %v, want %v",
err, errInterrupted)
}
want := "scan: interrupted after 0 files\n"
if got := stderr()[shown:]; got != want {
t.Errorf("stderr = %q, want %q", got, want)
}
if got := stdout(); got != "" {
t.Errorf("stdout = %q, want nothing (data only)", got)
}
assertNoSidecars(t, path)
if got := storedPaths(t, path); !slices.Equal(got, before) {
t.Errorf("records = %q after the interrupted scan, want %q",
got, before)
}
err = runScan(t.Context(), []string{dir}, walkCancelWorkers, false)
if err != nil {
t.Fatal(err)
}
got := storedPaths(t, path)
if slices.Contains(got, gone) || !slices.Contains(got, added) {
t.Errorf("records = %q after the next scan, want %q gone and %q "+
"added", got, gone, added)
}
}
// TestRunScanInterruptedMidHash interrupts the scan entrypoint part-way
// through its hash phase, after the database is open. It must return
// errInterrupted, release the lock, end stderr with its line counting
// every file the walk reached, write nothing to stdout, close the
// database out of WAL mode, and keep the records it hashed.
func TestRunScanInterruptedMidHash(t *testing.T) {
path := testDBPath(t)
t.Setenv(databaseEnv, path)
stdout := captureStdout(t)
stderr := captureStderr(t)
dir := buildWalkCancelTree(t)
err := runScan(newWalkClock(hashCancelAtDone), []string{dir},
walkCancelWorkers, false)
if !errors.Is(err, errInterrupted) {
t.Fatalf("runScan interrupted mid-hash = %v, want %v",
err, errInterrupted)
}
holdScanLock(t, path)
want := fmt.Sprintf("scan: interrupted after %d files\n", walkCancelFiles)
if got := stderr(); !strings.HasSuffix(got, want) {
t.Errorf("stderr = %q, want it to end with %q", got, want)
}
if got := stdout(); got != "" {
t.Errorf("stdout = %q, want nothing (data only)", got)
}
assertNoSidecars(t, path)
db, err := openReportDatabase(t.Context(), path)
if err != nil {
t.Fatal(err)
}
defer func() { _ = db.Close() }()
// A plain close also removes the sidecars, but leaves WAL mode on.
var mode string
err = db.QueryRowContext(t.Context(), "PRAGMA journal_mode").Scan(&mode)
if err != nil {
t.Fatal(err)
}
if mode != "delete" {
t.Errorf("journal mode = %q after the interrupted scan, want %q",
mode, "delete")
}
kept := len(dbRecords(t, db))
if kept == 0 || kept >= walkCancelFiles {
t.Errorf("%d records after the interrupted scan, want those it "+
"hashed: some but not all of the %d files", kept, walkCancelFiles)
}
}
// TestInterruptContextCatchesSIGTERM sends SIGTERM to the test process
// while the scan's handler is installed, and checks that it cancels the
// scan's context.
//
//nolint:paralleltest // signals the whole process: must not run beside a scan
func TestInterruptContextCatchesSIGTERM(t *testing.T) {
// Caught here as well, so that a handler that misses SIGTERM fails
// this test instead of ending the test process.
caught := make(chan os.Signal, 1)
signal.Notify(caught, syscall.SIGTERM)
defer signal.Stop(caught)
ctx, stop := interruptContext(t.Context())
defer stop()
err := syscall.Kill(os.Getpid(), syscall.SIGTERM)
if err != nil {
t.Fatal(err)
}
select {
case <-ctx.Done():
case <-time.After(poolUnwind):
t.Fatal("SIGTERM did not cancel the scan's context")
}
}
// TestCommitFullBatchKeepsFailedBatch checks that a full batch whose
// commit fails, as it does once the scan is interrupted, stays in the
// batch, so that syncScan's final commit saves it.
func TestCommitFullBatchKeepsFailedBatch(t *testing.T) {
t.Parallel()
s := &scanState{db: openTestDB(t)}
for i := range updateBatchSize {
s.batch = append(s.batch, scanRec{path: "/f" + strconv.Itoa(i)})
}
err := s.commitFullBatch(cancelledContext(t))
if !errors.Is(err, context.Canceled) {
t.Fatalf("commitFullBatch on a cancelled context = %v, want %v",
err, context.Canceled)
}
if len(s.batch) != updateBatchSize {
t.Errorf("batch holds %d records after the failed commit, want %d",
len(s.batch), updateBatchSize)
}
}
// drainClosed counts the values received from ch until it closes, // drainClosed counts the values received from ch until it closes,
// failing the test if it does not close within poolUnwind. A pool that // failing the test if it does not close within poolUnwind. A pool that
// ignored its cancellation leaves its channel open with its goroutines // ignored its cancellation leaves its channel open with its goroutines
@@ -596,49 +328,20 @@ func TestSendEventAbandonsBlockedSend(t *testing.T) {
awaitReturn(t, done, "sendEvent") awaitReturn(t, done, "sendEvent")
} }
// TestWalkOneDirStopsWhenCancelled checks that a cancelled scan stops
// reading a directory instead of going through the rest of its
// entries. A walk that kept going would return the subdirectory below
// to descend into. Unlike a file event, that return is not a send the
// cancellation can abandon, so the test catches the regression every
// time.
func TestWalkOneDirStopsWhenCancelled(t *testing.T) {
t.Parallel()
dir := t.TempDir()
err := os.Mkdir(filepath.Join(dir, "sub"), 0o750)
if err != nil {
t.Fatal(err)
}
// Unbuffered and unread: on a cancelled scan every send gives up.
events := make(chan walkEvent)
subs := walkOneDir(cancelledContext(t), dirJob{path: dir}, false, events)
if len(subs) != 0 {
t.Errorf("cancelled walkOneDir returned %+v to descend into, "+
"want none", subs)
}
}
// TestWalkWorkersDropQueuedDirs checks that cancelled walk workers keep // TestWalkWorkersDropQueuedDirs checks that cancelled walk workers keep
// reading jobs and drop the directories rather than stopping their // reading jobs and drop the directories rather than stopping their
// read: the range over jobs has to run out for the pool to tear down // read: the range over jobs has to run out for the pool to tear down
// and close its event stream. The queued directory does not exist, so // and close its event stream.
// a worker that walked it anyway would send a warning before
// walkOneDir's own cancellation check could stop it. On a cancelled
// scan that send delivers or gives up at random, so with 64 jobs
// queued the regression has a one in 2^64 chance of passing.
func TestWalkWorkersDropQueuedDirs(t *testing.T) { func TestWalkWorkersDropQueuedDirs(t *testing.T) {
t.Parallel() t.Parallel()
missing := filepath.Join(t.TempDir(), "missing") dir := t.TempDir()
writeEmptyFiles(t, dir, walkCancelFilesPerDir)
jobs, _, events := startWalkWorkers(cancelledContext(t), 2, false) jobs, _, events := startWalkWorkers(cancelledContext(t), 2, false)
for range 64 { for range 4 {
jobs <- dirJob{path: missing} jobs <- dirJob{path: dir}
} }
close(jobs) close(jobs)
@@ -709,11 +412,7 @@ func TestDispatchDirsClosesJobsWhenCancelled(t *testing.T) {
// TestFeedHashJobsClosesJobsWhenCancelled checks that the hash feeder // TestFeedHashJobsClosesJobsWhenCancelled checks that the hash feeder
// abandons the runs it has not queued yet and still closes the job // abandons the runs it has not queued yet and still closes the job
// channel, which is what lets the workers' range terminate. The // channel, which is what lets the workers' range terminate.
// receive on jobs below is not bounded: a feeder that returned without
// closing jobs would leave that receive with no sender and no close, so
// this regression is caught by the test binary's timeout rather than by
// a bounded assertion.
func TestFeedHashJobsClosesJobsWhenCancelled(t *testing.T) { func TestFeedHashJobsClosesJobsWhenCancelled(t *testing.T) {
t.Parallel() t.Parallel()
@@ -739,84 +438,41 @@ func TestFeedHashJobsClosesJobsWhenCancelled(t *testing.T) {
// TestHashWorkerDropsQueuedRuns checks that a cancelled hash worker // TestHashWorkerDropsQueuedRuns checks that a cancelled hash worker
// keeps reading jobs and drops the runs rather than reading files // keeps reading jobs and drops the runs rather than reading files
// nobody wants the hashes of — while still letting the range run out // nobody wants the hashes of — while still letting the range run out
// so the pool tears down. The hash function records that it was // so the pool tears down. The queued run names a file that does not
// called, so a worker that hashed the queued run anyway is caught // exist, so a worker that hashed it anyway would produce a result.
// every time.
func TestHashWorkerDropsQueuedRuns(t *testing.T) { func TestHashWorkerDropsQueuedRuns(t *testing.T) {
t.Parallel() t.Parallel()
done := make(chan struct{}) done := make(chan struct{})
jobs := make(chan []fileRec, 1) jobs := make(chan []fileRec, 1)
results := make(chan hashResult) results := make(chan hashResult, 1)
jobs <- []fileRec{{path: filepath.Join(t.TempDir(), "missing"), size: 1}} run := []fileRec{{path: filepath.Join(t.TempDir(), "missing"), size: 1}}
jobs <- run
close(jobs) close(jobs)
var hashed atomic.Bool
hash := func(path string, size int64) (string, string, string, error) {
hashed.Store(true)
return hashSignature(path, size)
}
go func() { go func() {
defer close(done) defer close(done)
hashWorker(cancelledContext(t), jobs, results, hash) hashWorker(cancelledContext(t), jobs, results, hashSignature)
}() }()
awaitReturn(t, done, "hashWorker") awaitReturn(t, done, "hashWorker")
if hashed.Load() { select {
t.Error("cancelled hash worker hashed the queued run, want it dropped") case r := <-results:
t.Errorf("cancelled hash worker produced %+v, want the run dropped",
r)
default:
} }
} }
// TestHashWorkerAbandonsBlockedSend checks that a hash worker with a
// result to deliver and nobody to deliver it to leaves once the scan
// is cancelled, instead of holding the pool open. The scan tests do
// not catch this: stop drains results, which frees a parked worker
// anyway.
func TestHashWorkerAbandonsBlockedSend(t *testing.T) {
t.Parallel()
ctx, cancel := context.WithCancel(t.Context())
defer cancel()
done := make(chan struct{})
jobs := make(chan []fileRec, 1)
// Unbuffered and unread, with jobs left open: the worker's only way
// out is the cancellation case beside its send.
results := make(chan hashResult)
jobs <- []fileRec{{path: filepath.Join(t.TempDir(), "missing"), size: 1}}
// The scan is cancelled while the worker hashes, so the worker has
// already passed the check that drops queued runs.
hash := func(path string, size int64) (string, string, string, error) {
cancel()
return hashSignature(path, size)
}
go func() {
defer close(done)
hashWorker(ctx, jobs, results, hash)
}()
awaitReturn(t, done, "hashWorker")
}
// TestHashPhaseCancelledReturnsContextError checks the result loop's // TestHashPhaseCancelledReturnsContextError checks the result loop's
// own exit: with the pool cancelled, no result will ever arrive, and // own exit: with the pool cancelled, no result will ever arrive, and
// the loop must leave through the cancellation rather than wait for a // the loop must leave through the cancellation rather than wait for a
// receive that cannot happen. This call is not bounded by poolUnwind: a // receive that cannot happen.
// loop that dropped its cancellation case would block on that receive,
// so the regression surfaces as the test binary's timeout rather than
// as a bounded assertion.
func TestHashPhaseCancelledReturnsContextError(t *testing.T) { func TestHashPhaseCancelledReturnsContextError(t *testing.T) {
t.Parallel() t.Parallel()
+14 -162
View File
@@ -6,7 +6,6 @@ import (
"errors" "errors"
"fmt" "fmt"
"io/fs" "io/fs"
"net/url"
"os" "os"
"path/filepath" "path/filepath"
"slices" "slices"
@@ -52,12 +51,6 @@ CREATE TABLE files (
) WITHOUT ROWID ) WITHOUT ROWID
` `
// createIndexSQL indexes the records by signature, so report can have
// SQLite group them without sorting the whole table.
const createIndexSQL = `
CREATE INDEX files_signature ON files (size, head, tail, content)
`
// upsertSQL inserts one file record, replacing any existing record for // upsertSQL inserts one file record, replacing any existing record for
// the same path. // the same path.
const upsertSQL = ` const upsertSQL = `
@@ -108,19 +101,7 @@ const reportParams = "mode=ro" +
// openDB opens the SQLite database at path with the connection // openDB opens the SQLite database at path with the connection
// parameters params. It does not create or verify the schema. // parameters params. It does not create or verify the schema.
func openDB(path, params string) (*sql.DB, error) { func openDB(path, params string) (*sql.DB, error) {
// The path is escaped into a file: URI, so ?, # and % in it stay db, err := sql.Open("sqlite", "file:"+path+"?"+params)
// part of the file name. SQLite reads what follows file:// up to
// the next / as a host name, so an absolute path goes after an
// empty host (file:///abs) and a relative path goes without one
// (file:rel).
uri := url.URL{
Scheme: "file",
OmitHost: !filepath.IsAbs(path),
Path: path,
RawQuery: params,
}
db, err := sql.Open("sqlite", uri.String())
if err != nil { if err != nil {
return nil, fmt.Errorf("open database %s: %w", path, err) return nil, fmt.Errorf("open database %s: %w", path, err)
} }
@@ -232,12 +213,6 @@ func openReportDatabase(ctx context.Context,
} }
v, err := userVersion(ctx, db) v, err := userVersion(ctx, db)
if err == nil && v == 0 {
// An empty database passes this check and fails the version
// check below.
err = checkUnversioned(ctx, db)
}
if err != nil { if err != nil {
_ = db.Close() _ = db.Close()
@@ -264,11 +239,6 @@ func initSchema(ctx context.Context, db *sql.DB) error {
switch v { switch v {
case 0: case 0:
err = checkUnversioned(ctx, db)
if err != nil {
return err
}
return createSchema(ctx, db) return createSchema(ctx, db)
case schemaVersion: case schemaVersion:
return nil return nil
@@ -278,65 +248,20 @@ func initSchema(ctx context.Context, db *sql.DB) error {
} }
} }
// checkUnversioned checks a database at user_version 0 before it is
// taken for an empty one. createSchema creates the files table and
// sets the version together, so a files table at version 0 was made by
// something else. Adopting it could corrupt unrelated data, so that is
// a schema-version error telling the operator to remove the file and
// rescan.
func checkUnversioned(ctx context.Context, db *sql.DB) error {
var name string
err := db.QueryRowContext(ctx,
"SELECT name FROM sqlite_master "+
"WHERE type = 'table' AND name = 'files'").Scan(&name)
switch {
case err == nil:
return fmt.Errorf(
"has a files table but no schema version; "+
"remove the file and rescan: %w", errSchemaVersion)
case errors.Is(err, sql.ErrNoRows):
return nil
default:
return fmt.Errorf("check for files table: %w", err)
}
}
// createSchema applies the schema to a fresh database and stamps the // createSchema applies the schema to a fresh database and stamps the
// schema version in one transaction, so a creation stopped partway, by // schema version.
// an interrupt or an error, leaves an empty database the next scan
// sets up, never a files table at version 0, which checkUnversioned
// refuses.
func createSchema(ctx context.Context, db *sql.DB) error { func createSchema(ctx context.Context, db *sql.DB) error {
tx, err := db.BeginTx(ctx, nil) _, err := db.ExecContext(ctx, createTableSQL)
if err != nil { if err != nil {
return fmt.Errorf("create schema: %w", err) return fmt.Errorf("create schema: %w", err)
} }
defer func() { _ = tx.Rollback() }() _, err = db.ExecContext(ctx,
_, err = tx.ExecContext(ctx, createTableSQL)
if err != nil {
return fmt.Errorf("create schema: %w", err)
}
_, err = tx.ExecContext(ctx, createIndexSQL)
if err != nil {
return fmt.Errorf("create schema: %w", err)
}
_, err = tx.ExecContext(ctx,
"PRAGMA user_version = "+strconv.Itoa(schemaVersion)) "PRAGMA user_version = "+strconv.Itoa(schemaVersion))
if err != nil { if err != nil {
return fmt.Errorf("set schema version: %w", err) return fmt.Errorf("set schema version: %w", err)
} }
err = tx.Commit()
if err != nil {
return fmt.Errorf("create schema: %w", err)
}
return nil return nil
} }
@@ -352,18 +277,18 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) {
return v, nil return v, nil
} }
// loadFileRows streams every record to fn in path order: byte order, // loadFileRows reads every record from the files table.
// which is the order of the primary key, so SQLite does not sort. func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
rows, err := db.QueryContext(ctx, rows, err := db.QueryContext(ctx,
"SELECT path, size, mtime, head, tail, content FROM files "+ "SELECT path, size, mtime, head, tail, content FROM files")
"ORDER BY path")
if err != nil { if err != nil {
return fmt.Errorf("read records: %w", err) return nil, fmt.Errorf("read records: %w", err)
} }
defer func() { _ = rows.Close() }() defer func() { _ = rows.Close() }()
var recs []scanRec
for rows.Next() { for rows.Next() {
var ( var (
path []byte path []byte
@@ -373,92 +298,19 @@ func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
&r.content) &r.content)
if err != nil { if err != nil {
return fmt.Errorf("read record: %w", err) return nil, fmt.Errorf("read record: %w", err)
} }
r.path = string(path) r.path = string(path)
fn(r) recs = append(recs, r)
} }
err = rows.Err() err = rows.Err()
if err != nil { if err != nil {
return fmt.Errorf("read records: %w", err) return nil, fmt.Errorf("read records: %w", err)
} }
return nil return recs, nil
}
// dupeRowsSQL selects every record in a duplicate group, with the
// group's first path. A group is the records with a content hash that
// share a size, head, tail, and content, when there are two or more of
// them. The rows come in report order: groups by size descending, then
// by first path, and each group's paths ascending.
const dupeRowsSQL = `
SELECT g.first, f.path, f.size
FROM files AS f
JOIN (
SELECT size, head, tail, content, MIN(path) AS first
FROM files
WHERE content <> ''
GROUP BY size, head, tail, content
HAVING COUNT(*) > 1
) AS g USING (size, head, tail, content)
ORDER BY f.size DESC, g.first, f.path
`
// loadDupeRows streams the rows of dupeRowsSQL to fn and returns the
// number of records in the database. The count and the rows are read
// in one transaction, so they agree while a scan is committing. An
// error from fn stops the reading and is returned as it is.
func loadDupeRows(ctx context.Context, db *sql.DB,
fn func(first, path string, size int64) error,
) (int, error) {
// Everything goes through tx: the report connection is the only
// one, so a query on db would wait for tx forever.
tx, err := db.BeginTx(ctx, &sql.TxOptions{ReadOnly: true})
if err != nil {
return 0, fmt.Errorf("read records: %w", err)
}
defer func() { _ = tx.Rollback() }()
var records int
err = tx.QueryRowContext(ctx, "SELECT COUNT(*) FROM files").Scan(&records)
if err != nil {
return 0, fmt.Errorf("read records: %w", err)
}
rows, err := tx.QueryContext(ctx, dupeRowsSQL)
if err != nil {
return 0, fmt.Errorf("read records: %w", err)
}
defer func() { _ = rows.Close() }()
for rows.Next() {
var (
first, path []byte
size int64
)
err = rows.Scan(&first, &path, &size)
if err != nil {
return 0, fmt.Errorf("read record: %w", err)
}
err = fn(string(first), string(path), size)
if err != nil {
return 0, err
}
}
err = rows.Err()
if err != nil {
return 0, fmt.Errorf("read records: %w", err)
}
return records, nil
} }
// loadFileMeta streams every record's path, size, mtime, and whether // loadFileMeta streams every record's path, size, mtime, and whether
+26 -89
View File
@@ -73,78 +73,9 @@ func TestOpenScanDatabaseCreates(t *testing.T) {
defer func() { _ = db.Close() }() defer func() { _ = db.Close() }()
if recs := dbRecords(t, db); len(recs) != 0 { recs, err := loadFileRows(t.Context(), db)
t.Fatalf("records = %v, want none", recs) if err != nil || len(recs) != 0 {
} t.Fatalf("loadFileRows = %v, %v; want empty, nil", recs, err)
}
func TestOpenDatabaseUnversionedForeign(t *testing.T) {
t.Parallel()
// A database that has a files table but user_version 0, written by
// some other tool. report, trees and scan must refuse it with the
// schema-version error, not adopt it and not emit a raw SQLite
// "table files already exists".
path := testDBPath(t)
db, err := sql.Open("sqlite", path)
if err != nil {
t.Fatal(err)
}
_, err = db.ExecContext(t.Context(), "CREATE TABLE files (x INTEGER)")
if err != nil {
t.Fatal(err)
}
_ = db.Close()
_, err = openReportDatabase(t.Context(), path)
if !errors.Is(err, errSchemaVersion) ||
!strings.Contains(err.Error(), "remove the file and rescan") {
t.Fatalf("report: err = %v, want errSchemaVersion telling the "+
"operator to remove the file and rescan", err)
}
_, err = openScanDatabase(t.Context(), path)
if !errors.Is(err, errSchemaVersion) ||
!strings.Contains(err.Error(), "remove the file and rescan") {
t.Fatalf("scan: err = %v, want errSchemaVersion telling the "+
"operator to remove the file and rescan", err)
}
}
func TestSchemaCreationStoppedPartway(t *testing.T) {
t.Parallel()
// A first scan stopped while creating the schema must leave a
// database the next scan accepts. max_page_count(2) leaves room for
// the files table but not its index, so schema creation fails right
// after CREATE TABLE, a point an interrupt could also stop it at.
path := testDBPath(t)
db, err := openDB(path, scanParams+"&_pragma=max_page_count(2)")
if err != nil {
t.Fatal(err)
}
err = initSchema(t.Context(), db)
_ = db.Close()
if err == nil {
t.Fatal("initSchema with no room for the index succeeded")
}
db, err = openScanDatabase(t.Context(), path)
if err != nil {
t.Fatalf("next scan: %v", err)
}
defer func() { _ = db.Close() }()
v, err := userVersion(t.Context(), db)
if err != nil || v != schemaVersion {
t.Fatalf("userVersion = %d, %v; want %d, nil", v, err, schemaVersion)
} }
} }
@@ -157,11 +88,9 @@ func TestOpenReportDatabaseMissing(t *testing.T) {
} }
} }
func TestOpenDatabaseVersionMismatch(t *testing.T) { func TestOpenReportDatabaseVersionMismatch(t *testing.T) {
t.Parallel() t.Parallel()
// A database stamped with a schema version other than 0 and
// schemaVersion. report, trees and scan must all refuse it.
path := testDBPath(t) path := testDBPath(t)
db, err := openScanDatabase(t.Context(), path) db, err := openScanDatabase(t.Context(), path)
@@ -178,12 +107,7 @@ func TestOpenDatabaseVersionMismatch(t *testing.T) {
_, err = openReportDatabase(t.Context(), path) _, err = openReportDatabase(t.Context(), path)
if !errors.Is(err, errSchemaVersion) { if !errors.Is(err, errSchemaVersion) {
t.Fatalf("report: err = %v, want errSchemaVersion", err) t.Fatalf("err = %v, want errSchemaVersion", err)
}
_, err = openScanDatabase(t.Context(), path)
if !errors.Is(err, errSchemaVersion) {
t.Fatalf("scan: err = %v, want errSchemaVersion", err)
} }
} }
@@ -244,7 +168,7 @@ func TestCloseScanDatabaseWhileReportOpen(t *testing.T) {
defer func() { _ = reportDB.Close() }() defer func() { _ = reportDB.Close() }()
err = loadFileRows(t.Context(), reportDB, func(scanRec) {}) _, err = loadFileRows(t.Context(), reportDB)
if err != nil { if err != nil {
t.Fatalf("loadFileRows: %v", err) t.Fatalf("loadFileRows: %v", err)
} }
@@ -272,8 +196,15 @@ func TestApplyChangesRoundTrip(t *testing.T) {
t.Fatalf("applyChanges: %v", err) t.Fatalf("applyChanges: %v", err)
} }
// The records come back in path order, which is the order of recs. got, err := loadFileRows(t.Context(), db)
got := dbRecords(t, db) if err != nil {
t.Fatal(err)
}
slices.SortFunc(got, func(a, b scanRec) int {
return strings.Compare(a.path, b.path)
})
if !slices.Equal(got, recs) { if !slices.Equal(got, recs) {
t.Fatalf("rows = %+v, want %+v", got, recs) t.Fatalf("rows = %+v, want %+v", got, recs)
} }
@@ -290,7 +221,11 @@ func TestApplyChangesRoundTrip(t *testing.T) {
t.Fatalf("applyChanges: %v", err) t.Fatalf("applyChanges: %v", err)
} }
got = dbRecords(t, db) got, err = loadFileRows(t.Context(), db)
if err != nil {
t.Fatal(err)
}
if len(got) != 1 || got[0] != upd { if len(got) != 1 || got[0] != upd {
t.Fatalf("rows = %+v, want just %+v", got, upd) t.Fatalf("rows = %+v, want just %+v", got, upd)
} }
@@ -319,8 +254,9 @@ func TestApplyChangesBatching(t *testing.T) {
t.Fatalf("applyChanges: %v", err) t.Fatalf("applyChanges: %v", err)
} }
if got := dbRecords(t, db); len(got) != n { got, err := loadFileRows(t.Context(), db)
t.Fatalf("records = %d, want %d", len(got), n) if err != nil || len(got) != n {
t.Fatalf("loadFileRows = %d rows, %v; want %d", len(got), err, n)
} }
deletes := make([]string, 0, n) deletes := make([]string, 0, n)
@@ -334,7 +270,8 @@ func TestApplyChangesBatching(t *testing.T) {
t.Fatalf("applyChanges deletes: %v", err) t.Fatalf("applyChanges deletes: %v", err)
} }
if got := dbRecords(t, db); len(got) != 0 { got, err = loadFileRows(t.Context(), db)
t.Fatalf("records = %d, want 0", len(got)) if err != nil || len(got) != 0 {
t.Fatalf("loadFileRows = %d rows, %v; want 0", len(got), err)
} }
} }
+16 -61
View File
@@ -15,7 +15,6 @@
// sfdupes scan [--workers N] [-x] PATH... // sfdupes scan [--workers N] [-x] PATH...
// sfdupes report > dupes.tsv // sfdupes report > dupes.tsv
// sfdupes trees > dupetrees.tsv // sfdupes trees > dupetrees.tsv
// sfdupes --version
// //
// See README.md for the complete specification. // See README.md for the complete specification.
package main package main
@@ -51,10 +50,6 @@ const (
// cobra prints for it is the whole message. // cobra prints for it is the whole message.
var errNoSubcommand = errors.New("no subcommand") var errNoSubcommand = errors.New("no subcommand")
// errWorkersBelowOne is the usage error for a scan --workers value
// below 1.
var errWorkersBelowOne = errors.New("--workers must be at least 1")
// Version is the build version, injected at link time via -ldflags // Version is the build version, injected at link time via -ldflags
// (see the Makefile); "dev" for a plain go build. // (see the Makefile); "dev" for a plain go build.
// //
@@ -92,9 +87,6 @@ func run(args []string, stdout, stderr io.Writer) int {
switch { switch {
case err == nil: case err == nil:
return exitOK return exitOK
case errors.Is(err, errInterrupted):
// The interrupted scan has printed its own line.
return exitFatal
case errors.As(err, &fatal): case errors.As(err, &fatal):
// The command ran and failed: a runtime error, reported // The command ran and failed: a runtime error, reported
// without the usage text that a usage error gets. // without the usage text that a usage error gets.
@@ -102,35 +94,22 @@ func run(args []string, stdout, stderr io.Writer) int {
return exitFatal return exitFatal
default: default:
// A usage error, which cobra has already reported on stderr. // A usage error: cobra has already printed the message and
// the usage text.
return exitUsage return exitUsage
} }
} }
// newRootCommand builds the command tree. Everything on stdout is // newRootCommand builds the command tree. Everything on stdout is
// machine-readable data, the version line included; all human-facing // machine-readable data; all human-facing output (help, usage, errors)
// output (help, usage, errors) goes to stderr. // goes to stderr.
func newRootCommand(stdout, stderr io.Writer) *cobra.Command { func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
var showVersion bool
printVersion := runE(func(context.Context, []string) error {
_, err := fmt.Fprintf(stdout, "sfdupes %s\n", Version)
if err != nil {
return fmt.Errorf("write stdout: %w", err)
}
return nil
})
root := &cobra.Command{ root := &cobra.Command{
Use: "sfdupes", Use: "sfdupes",
Short: "Find candidate duplicate files by size and head/tail/content SHA-256", Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
Args: cobra.NoArgs, Version: Version,
RunE: func(cmd *cobra.Command, args []string) error { Args: cobra.NoArgs,
if showVersion { RunE: func(cmd *cobra.Command, _ []string) error {
return printVersion(cmd, args)
}
// A missing subcommand prints usage and exits 2: cobra // A missing subcommand prints usage and exits 2: cobra
// prints the usage text for the returned error, and run // prints the usage text for the returned error, and run
// maps everything that is not a fatal error to exit 2. // maps everything that is not a fatal error to exit 2.
@@ -143,11 +122,6 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
root.SetErr(stderr) root.SetErr(stderr)
root.CompletionOptions.DisableDefaultCmd = true root.CompletionOptions.DisableDefaultCmd = true
// Cobra's built-in version flag prints through the help writer,
// stderr; this one prints to stdout.
root.Flags().BoolVarP(&showVersion, "version", "v", false,
"print the version to stdout")
var ( var (
scanWorkers int scanWorkers int
scanOneFS bool scanOneFS bool
@@ -157,13 +131,7 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
Use: cmdScan + " [--workers N] [-x] PATH...", Use: cmdScan + " [--workers N] [-x] PATH...",
Short: "Walk trees and synchronize the scan database", Short: "Walk trees and synchronize the scan database",
Args: cobra.MinimumNArgs(1), Args: cobra.MinimumNArgs(1),
PreRunE: func(cmd *cobra.Command, _ []string) error {
return checkScanWorkers(cmd, scanWorkers)
},
RunE: runE(func(ctx context.Context, args []string) error { RunE: runE(func(ctx context.Context, args []string) error {
ctx, stop := interruptContext(ctx)
defer stop()
return runScan(ctx, args, scanWorkers, scanOneFS) return runScan(ctx, args, scanWorkers, scanOneFS)
}), }),
} }
@@ -195,26 +163,13 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
return root return root
} }
// checkScanWorkers rejects a scan --workers value below 1. That is a // runE adapts a subcommand implementation to cobra's RunE. Cobra
// usage error reported in one line: cobra prints the returned message // prints the error and the command's usage text for every error RunE
// without the usage text, and run exits 2. // returns, but a subcommand that ran and failed has no usage problem
func checkScanWorkers(cmd *cobra.Command, workers int) error { // to report: both are silenced here, and the error is marked fatal so
if workers >= 1 { // that run reports it on stderr and exits 1 rather than 2. The command's
return nil // context is handed to the implementation: cancelling it unwinds the
} // scan's worker pools.
cmd.SilenceUsage = true
return fmt.Errorf("%w, got %d", errWorkersBelowOne, workers)
}
// runE adapts a subcommand implementation, or the version print, to
// cobra's RunE. Cobra prints the error and the command's usage text for
// every error RunE returns, but a subcommand that ran and failed has no
// usage problem to report: both are silenced here, and the error is
// marked fatal so that run reports it on stderr and exits 1 rather than
// 2. The command's context is handed to the implementation: cancelling
// it unwinds the scan's worker pools.
func runE( func runE(
fn func(ctx context.Context, args []string) error, fn func(ctx context.Context, args []string) error,
) func(*cobra.Command, []string) error { ) func(*cobra.Command, []string) error {
+53 -296
View File
@@ -8,7 +8,6 @@ import (
"io/fs" "io/fs"
"os" "os"
"path/filepath" "path/filepath"
"slices"
"strconv" "strconv"
"strings" "strings"
"testing" "testing"
@@ -85,41 +84,21 @@ func makeReadOnly(t *testing.T, path string) {
// captureStderr redirects os.Stderr to a file for the rest of the test // captureStderr redirects os.Stderr to a file for the rest of the test
// and returns a function reading back everything written to it. scan // and returns a function reading back everything written to it. scan
// writes its warnings and summary, and report and trees their // writes its warnings and summary straight to os.Stderr, not to the
// summaries, straight to os.Stderr, not to the stderr writer run is // stderr writer run is given.
// given.
func captureStderr(t *testing.T) func() string { func captureStderr(t *testing.T) func() string {
t.Helper() t.Helper()
return capture(t, &os.Stderr) f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
}
// captureStdout does for os.Stdout what captureStderr does for
// os.Stderr. scan is never given run's stdout writer, so anything it
// printed would go straight to os.Stdout. The scan tests pass os.Stdout
// as run's stdout too, so the one capture sees both.
func captureStdout(t *testing.T) func() string {
t.Helper()
return capture(t, &os.Stdout)
}
// capture redirects *std, which is os.Stdout or os.Stderr, to a file
// for the rest of the test and returns a function reading back
// everything written to it.
func capture(t *testing.T, std **os.File) func() string {
t.Helper()
f, err := os.Create(filepath.Join(t.TempDir(), "output"))
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
saved := *std saved := os.Stderr
*std = f os.Stderr = f
t.Cleanup(func() { t.Cleanup(func() {
*std = saved os.Stderr = saved
_ = f.Close() _ = f.Close()
}) })
@@ -204,33 +183,30 @@ func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
// sidecar check is evidence of the close only for scan: report and // sidecar check is evidence of the close only for scan: report and
// trees only read a database that is out of WAL mode, which leaves // trees only read a database that is out of WAL mode, which leaves
// nothing on disk whether they close it or not. // nothing on disk whether they close it or not.
// cases := map[string][]string{
// The subcommands come from the command tree, so a new one is cmdScan: {cmdScan},
// checked too: one wired with a bare RunE instead of runE reports cmdReport: {cmdReport},
// its failure as a usage error, exit 2 with the usage text. cmdTrees: {cmdTrees},
for _, cmd := range newRootCommand(io.Discard, io.Discard).Commands() { }
name := cmd.Name()
for name, args := range cases {
t.Run(name, func(t *testing.T) { t.Run(name, func(t *testing.T) {
path := brokenDatabase(t) path := brokenDatabase(t)
t.Setenv(databaseEnv, path) t.Setenv(databaseEnv, path)
args := []string{name}
if name == cmdScan { if name == cmdScan {
args = append(args, t.TempDir()) args = append(args, t.TempDir())
} }
stdout := captureStdout(t) var stdout, stderr bytes.Buffer
var stderr bytes.Buffer code := run(args, &stdout, &stderr)
code := run(args, os.Stdout, &stderr)
if code != exitFatal { if code != exitFatal {
t.Errorf("run(%v) = %d, want %d", args, code, exitFatal) t.Errorf("run(%v) = %d, want %d", args, code, exitFatal)
} }
assertNoSidecars(t, path) assertNoSidecars(t, path)
assertFatalOutput(t, stderr.String(), stdout()) assertFatalOutput(t, stderr.String(), stdout.String())
// Proof that the failure happened after the open: only a // Proof that the failure happened after the open: only a
// query against the opened database can report this. // query against the opened database can report this.
@@ -242,24 +218,22 @@ func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
} }
} }
func TestRunNonexistentPathIsFatalNotUsage(t *testing.T) { func TestRunMissingOperandIsFatalNotUsage(t *testing.T) {
// README §Error handling: a PATH operand that does not exist is a // README §Error handling: a PATH operand that does not exist is a
// fatal error (1), not a usage error (2) — and a runtime failure // fatal error (1), not a usage error (2) — and a runtime failure
// must not dump the usage text. // must not dump the usage text.
t.Setenv(databaseEnv, testDBPath(t)) t.Setenv(databaseEnv, testDBPath(t))
stdout := captureStdout(t) var stdout, stderr bytes.Buffer
var stderr bytes.Buffer
missing := filepath.Join(t.TempDir(), "nope") missing := filepath.Join(t.TempDir(), "nope")
code := run([]string{cmdScan, missing}, os.Stdout, &stderr) code := run([]string{cmdScan, missing}, &stdout, &stderr)
if code != exitFatal { if code != exitFatal {
t.Errorf("run(scan %s) = %d, want %d", missing, code, exitFatal) t.Errorf("run(scan %s) = %d, want %d", missing, code, exitFatal)
} }
assertFatalOutput(t, stderr.String(), stdout()) assertFatalOutput(t, stderr.String(), stdout.String())
} }
// assertFatalOutput checks that a fatal error was reported the way // assertFatalOutput checks that a fatal error was reported the way
@@ -283,34 +257,6 @@ func assertFatalOutput(t *testing.T, stderr, stdout string) {
} }
} }
func TestRunMissingDatabaseIsFatal(t *testing.T) {
// README §Database: report and trees need an existing database; a
// missing one exits 1 with a message telling the user to run scan.
for _, name := range []string{cmdReport, cmdTrees} {
t.Run(name, func(t *testing.T) {
path := testDBPath(t)
t.Setenv(databaseEnv, path)
var stdout, stderr bytes.Buffer
code := run([]string{name}, &stdout, &stderr)
if code != exitFatal {
t.Errorf("run(%s) = %d, want %d", name, code, exitFatal)
}
want := "sfdupes: " + path + ": no database (run \"sfdupes " +
"scan\" first, or set " + databaseEnv + ")\n"
if got := stderr.String(); got != want {
t.Errorf("stderr = %q, want %q", got, want)
}
if got := stdout.String(); got != "" {
t.Errorf("stdout = %q, want nothing (data only)", got)
}
})
}
}
func TestRunUsageErrors(t *testing.T) { func TestRunUsageErrors(t *testing.T) {
// Usage errors keep exiting 2 with cobra's own report on stderr. // Usage errors keep exiting 2 with cobra's own report on stderr.
cases := map[string]struct { cases := map[string]struct {
@@ -349,108 +295,34 @@ func TestRunUsageErrors(t *testing.T) {
} }
} }
func TestRunScanRejectsWorkersBelowOne(t *testing.T) { // TestRunHelpAndVersionSucceed checks that the two informational flags
// README §scan mode: --workers below 1 is a usage error reported in // exit 0 and keep their human-facing output on stderr.
// one line on stderr, before the scan opens the database. func TestRunHelpAndVersionSucceed(t *testing.T) {
for _, workers := range []string{"0", "-1"} {
t.Run(workers, func(t *testing.T) {
dbPath := testDBPath(t)
t.Setenv(databaseEnv, dbPath)
var stdout, stderr bytes.Buffer
args := []string{cmdScan, "--workers", workers, t.TempDir()}
code := run(args, &stdout, &stderr)
if code != exitUsage {
t.Errorf("run(%v) = %d, want %d", args, code, exitUsage)
}
want := "Error: --workers must be at least 1, got " + workers +
"\n"
if got := stderr.String(); got != want {
t.Errorf("stderr = %q, want %q", got, want)
}
if got := stdout.String(); got != "" {
t.Errorf("stdout = %q, want nothing (data only)", got)
}
_, err := os.Stat(dbPath)
if !errors.Is(err, fs.ErrNotExist) {
t.Errorf("stat %s: %v, want the database never created",
dbPath, err)
}
})
}
}
func TestRunHelp(t *testing.T) {
t.Parallel() t.Parallel()
// README §Subcommands: help goes to stderr, exits 0, and leaves assertHumanOutput(t, "--help")
// stdout empty. assertHumanOutput(t, "--version")
cases := [][]string{{"--help"}, {"-h"}, {cmdScan, "--help"}}
for _, args := range cases {
var stdout, stderr bytes.Buffer
code := run(args, &stdout, &stderr)
if code != exitOK {
t.Errorf("run(%v) = %d, want %d", args, code, exitOK)
}
if !strings.Contains(stderr.String(), usageMarker) {
t.Errorf("run(%v) stderr = %q, want the help text",
args, stderr.String())
}
if got := stdout.String(); got != "" {
t.Errorf("run(%v) stdout = %q, want nothing (data only)",
args, got)
}
}
} }
func TestRunVersion(t *testing.T) { // assertHumanOutput runs sfdupes with one informational flag and checks
t.Parallel() // that it succeeds with its output on stderr and stdout untouched
// (README design goal 4).
func assertHumanOutput(t *testing.T, arg string) {
t.Helper()
// README §Subcommands: the version is one line on stdout, with var stdout, stderr bytes.Buffer
// nothing on stderr, and exits 0.
for _, arg := range []string{"--version", "-v"} {
var stdout, stderr bytes.Buffer
code := run([]string{arg}, &stdout, &stderr) code := run([]string{arg}, &stdout, &stderr)
if code != exitOK { if code != exitOK {
t.Errorf("run(%s) = %d, want %d", arg, code, exitOK) t.Errorf("run(%s) = %d, want %d", arg, code, exitOK)
}
want := "sfdupes " + Version + "\n"
if got := stdout.String(); got != want {
t.Errorf("run(%s) stdout = %q, want %q", arg, got, want)
}
if got := stderr.String(); got != "" {
t.Errorf("run(%s) stderr = %q, want nothing", arg, got)
}
}
}
func TestRunVersionWriteFailureIsFatal(t *testing.T) {
t.Parallel()
// README §Error handling: a stdout write failure exits 1, reported
// in one line on stderr.
var stderr bytes.Buffer
code := run([]string{"--version"}, failingWriter{}, &stderr)
if code != exitFatal {
t.Errorf("run(--version) = %d, want %d", code, exitFatal)
} }
want := "sfdupes: write stdout: " + errWriteFailed.Error() + "\n" if stderr.Len() == 0 {
if got := stderr.String(); got != want { t.Errorf("run(%s) wrote nothing to stderr", arg)
t.Errorf("stderr = %q, want %q", got, want) }
if got := stdout.String(); got != "" {
t.Errorf("stdout = %q, want nothing (data only)", got)
} }
} }
@@ -488,16 +360,17 @@ func scanFixture(t *testing.T) []string {
func scanOK(t *testing.T, operands ...string) string { func scanOK(t *testing.T, operands ...string) string {
t.Helper() t.Helper()
stdout := captureStdout(t) var stdout bytes.Buffer
stderr := captureStderr(t) stderr := captureStderr(t)
code := run(append([]string{cmdScan}, operands...), os.Stdout, os.Stderr) code := run(append([]string{cmdScan}, operands...), &stdout, os.Stderr)
if code != exitOK { if code != exitOK {
t.Fatalf("run(scan %q) = %d, want %d; stderr: %s", t.Fatalf("run(scan %q) = %d, want %d; stderr: %s",
operands, code, exitOK, stderr()) operands, code, exitOK, stderr())
} }
if got := stdout(); got != "" { if got := stdout.String(); got != "" {
t.Errorf("scan stdout = %q, want nothing (data only)", got) t.Errorf("scan stdout = %q, want nothing (data only)", got)
} }
@@ -505,41 +378,10 @@ func scanOK(t *testing.T, operands ...string) string {
} }
func TestRunScanSucceedsDespiteWarnings(t *testing.T) { func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
// README §Error handling: a scan that skips a file it cannot read
// warns, counts the skip in its summary, and still exits 0, which
// scanOK checks along with the empty stdout.
if os.Geteuid() == 0 {
t.Skip("root ignores file permissions")
}
path := testDBPath(t) path := testDBPath(t)
t.Setenv(databaseEnv, path) t.Setenv(databaseEnv, path)
dir := t.TempDir() scanFixture(t)
writeFile(t, dir, "a.bin", pattern(1, 300))
// Same size as a.bin, so the scan reads it, and the read fails.
unreadable := writeFile(t, dir, "unreadable.bin", pattern(2, 300))
err := os.Chmod(unreadable, 0)
if err != nil {
t.Fatal(err)
}
stderr := scanOK(t, dir)
warning := "hash " + unreadable + ": open " + unreadable +
": permission denied\n"
if !strings.Contains(stderr, warning) {
t.Errorf("stderr = %q, want %q", stderr, warning)
}
summary := "scan: 1 files seen (1 added, 0 updated, 0 removed, " +
"0 unchanged), 1 skipped\n"
if !strings.Contains(stderr, summary) {
t.Errorf("stderr = %q, want %q", stderr, summary)
}
assertNoSidecars(t, path) assertNoSidecars(t, path)
} }
@@ -646,14 +488,12 @@ func TestRunReportSucceeds(t *testing.T) {
dupes := scanFixture(t) dupes := scanFixture(t)
var stdout bytes.Buffer var stdout, stderr bytes.Buffer
stderr := captureStderr(t) code := run([]string{cmdReport}, &stdout, &stderr)
code := run([]string{cmdReport}, &stdout, os.Stderr)
if code != exitOK { if code != exitOK {
t.Fatalf("run(report) = %d, want %d; stderr: %s", t.Fatalf("run(report) = %d, want %d; stderr: %s",
code, exitOK, stderr()) code, exitOK, stderr.String())
} }
want := "first\tdupe\tsize\n" + dupes[0] + "\t" + dupes[1] + "\t300\n" want := "first\tdupe\tsize\n" + dupes[0] + "\t" + dupes[1] + "\t300\n"
@@ -661,12 +501,6 @@ func TestRunReportSucceeds(t *testing.T) {
t.Errorf("stdout = %q, want %q", got, want) t.Errorf("stdout = %q, want %q", got, want)
} }
want = "report: 2 records read, 1 duplicate groups, 1 dupe files, " +
"300 B reclaimable\n"
if got := stderr(); got != want {
t.Errorf("stderr = %q, want %q", got, want)
}
assertNoSidecars(t, path) assertNoSidecars(t, path)
} }
@@ -676,14 +510,12 @@ func TestRunTreesSucceeds(t *testing.T) {
dupes := scanFixture(t) dupes := scanFixture(t)
var stdout bytes.Buffer var stdout, stderr bytes.Buffer
stderr := captureStderr(t) code := run([]string{cmdTrees}, &stdout, &stderr)
code := run([]string{cmdTrees}, &stdout, os.Stderr)
if code != exitOK { if code != exitOK {
t.Fatalf("run(trees) = %d, want %d; stderr: %s", t.Fatalf("run(trees) = %d, want %d; stderr: %s",
code, exitOK, stderr()) code, exitOK, stderr.String())
} }
// The two directories holding the duplicate pair are duplicate // The two directories holding the duplicate pair are duplicate
@@ -694,12 +526,6 @@ func TestRunTreesSucceeds(t *testing.T) {
t.Errorf("stdout = %q, want %q", got, want) t.Errorf("stdout = %q, want %q", got, want)
} }
want = "trees: 2 records read, 1 duplicate tree groups, 1 dupe trees, " +
"300 B reclaimable\n"
if got := stderr(); got != want {
t.Errorf("stderr = %q, want %q", got, want)
}
assertNoSidecars(t, path) assertNoSidecars(t, path)
} }
@@ -739,73 +565,6 @@ func TestRunReportsNeedOnlyReadAccess(t *testing.T) {
} }
} }
// assertRunsUseDatabase runs scan, then report and trees, against the
// database that SFDUPES_DATABASE names, the file name in dir. It fails
// unless the reports find the duplicate pair the scan recorded and dir
// then holds only that file and its lock file: nothing was created
// under a shortened name.
func assertRunsUseDatabase(t *testing.T, dir, name string) {
t.Helper()
dupes := scanFixture(t)
want := "first\tdupe\tsize\n" + dupes[0] + "\t" + dupes[1] + "\t300\n"
if got := runStdout(t, cmdReport); got != want {
t.Errorf("report stdout = %q, want %q", got, want)
}
want = "first\tdupe\tfiles\tsize\n" +
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) + "\t1\t300\n"
if got := runStdout(t, cmdTrees); got != want {
t.Errorf("trees stdout = %q, want %q", got, want)
}
entries, err := os.ReadDir(dir)
if err != nil {
t.Fatal(err)
}
got := make([]string, 0, len(entries))
for _, e := range entries {
got = append(got, e.Name())
}
if wantFiles := []string{name, name + ".lock"}; !slices.Equal(got, wantFiles) {
t.Errorf("%s holds %q, want %q", dir, got, wantFiles)
}
}
func TestRunDatabasePathUsedAsGiven(t *testing.T) {
// README §Database: the path names the database file exactly. In
// SQLite's connection string an unescaped ? or # would end the file
// name and % would start an escape, and a path starting with //
// could be read as a host name. %25 is a valid escape, so unescaped
// this name opens a file named a without any error.
const name = "a?b#c%25d e.sqlite"
t.Run("absolute", func(t *testing.T) {
dir := t.TempDir()
t.Setenv(databaseEnv, filepath.Join(dir, name))
assertRunsUseDatabase(t, dir, name)
})
t.Run("leading double slash", func(t *testing.T) {
dir := t.TempDir()
t.Setenv(databaseEnv, "/"+filepath.Join(dir, name))
assertRunsUseDatabase(t, dir, name)
})
t.Run("relative", func(t *testing.T) {
dir := t.TempDir()
t.Chdir(dir)
t.Setenv(databaseEnv, name)
assertRunsUseDatabase(t, dir, name)
})
}
// holdScanLock takes the lock on the database at path, as a running // holdScanLock takes the lock on the database at path, as a running
// scan does, and holds it until the test ends. It fails the test when // scan does, and holds it until the test ends. It fails the test when
// the lock is already held. // the lock is already held.
@@ -829,11 +588,9 @@ func TestRunSecondScanFails(t *testing.T) {
holdScanLock(t, path) holdScanLock(t, path)
stdout := captureStdout(t) var stdout, stderr bytes.Buffer
var stderr bytes.Buffer code := run([]string{cmdScan, t.TempDir()}, &stdout, &stderr)
code := run([]string{cmdScan, t.TempDir()}, os.Stdout, &stderr)
if code != exitFatal { if code != exitFatal {
t.Errorf("run(scan) = %d, want %d", code, exitFatal) t.Errorf("run(scan) = %d, want %d", code, exitFatal)
} }
@@ -844,7 +601,7 @@ func TestRunSecondScanFails(t *testing.T) {
t.Errorf("stderr = %q, want %q", got, want) t.Errorf("stderr = %q, want %q", got, want)
} }
if got := stdout(); got != "" { if got := stdout.String(); got != "" {
t.Errorf("stdout = %q, want nothing (data only)", got) t.Errorf("stdout = %q, want nothing (data only)", got)
} }
-5
View File
@@ -1,5 +0,0 @@
{
"devDependencies": {
"prettier": "3.8.1"
}
}
+2 -8
View File
@@ -141,20 +141,14 @@ func (p *progress) warnf(format string, args ...any) {
fmt.Fprintln(os.Stderr, msg) fmt.Fprintln(os.Stderr, msg)
} }
// finish terminates the pass's display. A bar whose pass stopped short // finish terminates the pass's display.
// of its total, as an interrupted one does, is left as last drawn; the
// library's Finish would fill it up.
func (p *progress) finish() { func (p *progress) finish() {
if p == nil { if p == nil {
return return
} }
if p.bar != nil { if p.bar != nil {
if p.total >= 0 && p.count < p.total { _ = p.bar.Finish()
_ = p.bar.Exit()
} else {
_ = p.bar.Finish()
}
fmt.Fprintln(os.Stderr) fmt.Fprintln(os.Stderr)
+6 -34
View File
@@ -144,46 +144,18 @@ func TestSpinnerShowsCountAfterBurst(t *testing.T) {
time.Sleep(spinnerIdle) time.Sleep(spinnerIdle)
if shown := lastFrame(stderr()); !strings.Contains(shown, "(50/-,") { // A terminal shows the last frame drawn. The library starts each
t.Errorf("terminal shows %q, want a count of 50", shown) // frame with a carriage return and erases the previous one with
} // spaces first.
}
// lastFrame returns what a terminal shows of the frames a bar drew: the
// last one. The library starts each frame with a carriage return and
// erases the previous one with spaces first.
func lastFrame(out string) string {
var shown string var shown string
for frame := range strings.SplitSeq(out, "\r") { for frame := range strings.SplitSeq(stderr(), "\r") {
if strings.TrimSpace(frame) != "" { if strings.TrimSpace(frame) != "" {
shown = frame shown = frame
} }
} }
return shown if !strings.Contains(shown, "(50/-,") {
} t.Errorf("terminal shows %q, want a count of 50", shown)
// TestBarStoppedShortKeepsCount checks that the terminal display of a
// pass that stops before its total, as an interrupted one does, is left
// as last drawn instead of being filled up.
//
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
func TestBarStoppedShortKeepsCount(t *testing.T) {
stderr := captureStderr(t)
p := &progress{
label: "hash", total: 10, start: time.Now(),
bar: newBar("hash", 10),
}
p.increment()
// Past the redraw limit, so the bar draws the next count.
time.Sleep(2 * barThrottle)
p.increment()
p.finish()
if shown := lastFrame(stderr()); !strings.Contains(shown, "(2/10,") {
t.Errorf("terminal shows %q, want a count of 2 of 10", shown)
} }
} }
+95 -34
View File
@@ -6,6 +6,7 @@ import (
"fmt" "fmt"
"io" "io"
"os" "os"
"slices"
"strings" "strings"
) )
@@ -28,22 +29,51 @@ type scanRec struct {
path string path string
} }
// runReport implements the report subcommand: it prints the file-level // loadRecords opens the database and reads every file record for the
// duplicates report as TSV on stdout. SQLite groups and orders the // report and trees subcommands. Any database problem — including a
// records, and each row is written as it is read, so no group is held // missing database — is fatal. The error is returned rather than
// in memory. It never touches the scanned filesystem; its only I/O is // exiting, so that the deferred close always runs; the database is
// the database (with SQLite's temporary sort file), stdout, and stderr. // closed before the caller formats its output, so it stays closed even
// Any database problem, including a missing database, is fatal. // if that output fails.
func runReport(ctx context.Context, stdout io.Writer) error { func loadRecords(ctx context.Context) ([]scanRec, error) {
dbPath := databasePath() dbPath := databasePath()
db, err := openReportDatabase(ctx, dbPath) db, err := openReportDatabase(ctx, dbPath)
if err != nil { if err != nil {
return err return nil, err
} }
defer func() { _ = db.Close() }() defer func() { _ = db.Close() }()
recs, err := loadFileRows(ctx, db)
if err != nil {
return nil, fmt.Errorf("database %s: %w", dbPath, err)
}
return recs, nil
}
// dupeGroup is one set of candidate-duplicate files: identical size,
// head hash, tail hash, and content hash. paths is sorted
// lexicographically; the first entry is the group's "first", the rest
// are dupes.
type dupeGroup struct {
size int64
paths []string
}
// runReport implements the report subcommand: it reads every record
// from the database and prints the file-level duplicates report as TSV
// on stdout. It never touches the scanned filesystem; its only I/O is
// the database, stdout, and stderr.
func runReport(ctx context.Context, stdout io.Writer) error {
recs, err := loadRecords(ctx)
if err != nil {
return err
}
dupes := collectDupeGroups(recs)
out := bufio.NewWriterSize(stdout, ioBufSize) out := bufio.NewWriterSize(stdout, ioBufSize)
_, err = fmt.Fprintln(out, "first\tdupe\tsize") _, err = fmt.Fprintln(out, "first\tdupe\tsize")
@@ -51,36 +81,21 @@ func runReport(ctx context.Context, stdout io.Writer) error {
return fmt.Errorf("write stdout: %w", err) return fmt.Errorf("write stdout: %w", err)
} }
var ( dupeFiles := 0
groups, dupeFiles int
reclaimable int64
writeErr error
)
records, err := loadDupeRows(ctx, db, var reclaimable int64
func(first, path string, size int64) error {
// A group's first path is its first row; every other
// path is a dupe.
if path == first {
groups++
return nil for _, g := range dupes {
for _, p := range g.paths[1:] {
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
escapePath(g.paths[0]), escapePath(p), g.size)
if err != nil {
return fmt.Errorf("write stdout: %w", err)
} }
_, writeErr = fmt.Fprintf(out, "%s\t%s\t%d\n",
escapePath(first), escapePath(path), size)
dupeFiles++ dupeFiles++
reclaimable += size reclaimable += g.size
}
return writeErr
})
if writeErr != nil {
return fmt.Errorf("write stdout: %w", writeErr)
}
if err != nil {
return fmt.Errorf("database %s: %w", dbPath, err)
} }
err = out.Flush() err = out.Flush()
@@ -91,11 +106,57 @@ func runReport(ctx context.Context, stdout io.Writer) error {
fmt.Fprintf(os.Stderr, fmt.Fprintf(os.Stderr,
"report: %d records read, %d duplicate groups, %d dupe files, "+ "report: %d records read, %d duplicate groups, %d dupe files, "+
"%s reclaimable\n", "%s reclaimable\n",
records, groups, dupeFiles, humanBytes(reclaimable)) len(recs), len(dupes), dupeFiles, humanBytes(reclaimable))
return nil return nil
} }
// collectDupeGroups groups records by signature and returns every group
// with two or more paths, each group's paths sorted lexicographically,
// groups ordered by size descending then by first path ascending.
func collectDupeGroups(recs []scanRec) []dupeGroup {
groups := make(map[fileSig][]string)
for _, r := range recs {
// A record without a content hash has unknown content and is
// never reported as a duplicate (README "Database").
if r.content == "" {
continue
}
k := fileSig{
size: r.size, head: r.head, tail: r.tail, content: r.content,
}
groups[k] = append(groups[k], r.path)
}
var dupes []dupeGroup
for k, paths := range groups {
if len(paths) < minGroupSize {
continue
}
slices.Sort(paths)
dupes = append(dupes, dupeGroup{size: k.size, paths: paths})
}
// Biggest reclaimable space first; ties broken by first path.
slices.SortFunc(dupes, func(a, b dupeGroup) int {
if a.size != b.size {
if a.size > b.size {
return -1
}
return 1
}
return strings.Compare(a.paths[0], b.paths[0])
})
return dupes
}
// escapePath returns a path as it is written in a report column (README // escapePath returns a path as it is written in a report column (README
// "Report output format"): a backslash, tab, newline or carriage return // "Report output format"): a backslash, tab, newline or carriage return
// becomes \\, \t, \n or \r, and every other byte is kept as it is. // becomes \\, \t, \n or \r, and every other byte is kept as it is.
+15 -144
View File
@@ -2,14 +2,10 @@ package main
import ( import (
"bytes" "bytes"
"database/sql"
"errors"
"fmt"
"io" "io"
"os" "os"
"path/filepath" "path/filepath"
"slices" "slices"
"strings"
"testing" "testing"
) )
@@ -51,53 +47,6 @@ func seedDatabase(t *testing.T, recs []scanRec) string {
return path return path
} }
// dupeGroup is one duplicate group as report reads it: the size, and
// the paths in report order, first path first.
type dupeGroup struct {
size int64
paths []string
}
// dupeGroups returns the duplicate groups report reads from db, in
// report order.
func dupeGroups(t *testing.T, db *sql.DB) []dupeGroup {
t.Helper()
var groups []dupeGroup
_, err := loadDupeRows(t.Context(), db,
func(first, path string, size int64) error {
if path == first {
groups = append(groups, dupeGroup{size: size})
}
g := &groups[len(groups)-1]
g.paths = append(g.paths, path)
return nil
})
if err != nil {
t.Fatal(err)
}
return groups
}
// dupeGroupsOf writes recs into a fresh database and returns the
// duplicate groups report reads from it.
func dupeGroupsOf(t *testing.T, recs []scanRec) []dupeGroup {
t.Helper()
db := openTestDB(t)
err := applyChanges(t.Context(), db, recs, nil, nil)
if err != nil {
t.Fatal(err)
}
return dupeGroups(t, db)
}
func TestRunReportEscapesPaths(t *testing.T) { func TestRunReportEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs())) t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
@@ -116,82 +65,6 @@ func TestRunReportEscapesPaths(t *testing.T) {
} }
} }
func TestReportStdoutFailsWhileReading(t *testing.T) {
// Each row holds two paths longer than dir, so the report is more
// than twice the stdout buffer and stdout fails while rows are
// still being read, not at the final flush.
dir := "/" + strings.Repeat("d", 4096)
recs := make([]scanRec, ioBufSize/len(dir))
for i := range recs {
recs[i] = scanRec{
size: 1, head: "h", tail: "t", content: "c",
path: fmt.Sprintf("%s/%d", dir, i),
}
}
t.Setenv(databaseEnv, seedDatabase(t, recs))
err := runReport(t.Context(), failingWriter{})
if !errors.Is(err, errWriteFailed) ||
!strings.HasPrefix(err.Error(), "write stdout: ") {
t.Errorf("error = %v, want write stdout: %v", err, errWriteFailed)
}
}
func TestRunReportsIgnoreInsertionOrder(t *testing.T) {
// README §Constraints: identical database contents give identical
// output, whatever order the records were inserted in.
recs := append(smokeTreeRecs(), awkwardPairRecs()...)
recs = append(recs,
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/2"},
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/1"},
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/2"},
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/1"},
scanRec{size: 50, path: "/x/unhashed"},
)
reversed := slices.Clone(recs)
slices.Reverse(reversed)
for _, name := range []string{cmdReport, cmdTrees} {
t.Run(name, func(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, recs))
forward := runStdout(t, name)
t.Setenv(databaseEnv, seedDatabase(t, reversed))
backward := runStdout(t, name)
if strings.Count(forward, "\n") < 3 {
t.Errorf("stdout = %q, want at least two rows", forward)
}
if forward != backward {
t.Errorf("stdout depends on insertion order: %q vs %q",
forward, backward)
}
})
}
}
// runStdout runs the subcommand name and returns its stdout, failing
// the test unless it succeeds.
func runStdout(t *testing.T, name string) string {
t.Helper()
var stdout, stderr bytes.Buffer
code := run([]string{name}, &stdout, &stderr)
if code != exitOK {
t.Fatalf("run(%s) = %d, want %d; stderr: %s",
name, code, exitOK, stderr.String())
}
return stdout.String()
}
func TestEscapePath(t *testing.T) { func TestEscapePath(t *testing.T) {
t.Parallel() t.Parallel()
@@ -248,7 +121,7 @@ func TestWarnfEscapes(t *testing.T) {
} }
} }
func TestDupeGroups(t *testing.T) { func TestCollectDupeGroups(t *testing.T) {
t.Parallel() t.Parallel()
recs := []scanRec{ recs := []scanRec{
@@ -263,7 +136,7 @@ func TestDupeGroups(t *testing.T) {
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"}, {size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
} }
groups := dupeGroupsOf(t, recs) groups := collectDupeGroups(recs)
if len(groups) != 2 { if len(groups) != 2 {
t.Fatalf("len(groups) = %d, want 2", len(groups)) t.Fatalf("len(groups) = %d, want 2", len(groups))
} }
@@ -281,7 +154,7 @@ func TestDupeGroups(t *testing.T) {
} }
} }
func TestDupeGroupsContentSeparates(t *testing.T) { func TestCollectDupeGroupsContentSeparates(t *testing.T) {
t.Parallel() t.Parallel()
// Same size, head, and tail, but different content hashes: the final // Same size, head, and tail, but different content hashes: the final
@@ -296,7 +169,7 @@ func TestDupeGroupsContentSeparates(t *testing.T) {
{size: 100, head: "h", tail: "t", path: "/e"}, {size: 100, head: "h", tail: "t", path: "/e"},
} }
groups := dupeGroupsOf(t, recs) groups := collectDupeGroups(recs)
if len(groups) != 1 { if len(groups) != 1 {
t.Fatalf("len(groups) = %d, want 1 (only the matching content)", t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
len(groups)) len(groups))
@@ -307,7 +180,7 @@ func TestDupeGroupsContentSeparates(t *testing.T) {
} }
} }
func TestDupeGroupsMtimeExcluded(t *testing.T) { func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
t.Parallel() t.Parallel()
// mtime is informational only; records differing only in mtime // mtime is informational only; records differing only in mtime
@@ -317,25 +190,23 @@ func TestDupeGroupsMtimeExcluded(t *testing.T) {
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"}, {size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
} }
groups := dupeGroupsOf(t, recs) groups := collectDupeGroups(recs)
if len(groups) != 1 { if len(groups) != 1 {
t.Fatalf("len(groups) = %d, want 1", len(groups)) t.Fatalf("len(groups) = %d, want 1", len(groups))
} }
} }
func TestDupeGroupsTieBreak(t *testing.T) { func TestCollectDupeGroupsTieBreak(t *testing.T) {
t.Parallel() t.Parallel()
// The hashes sort opposite to the first paths, so ordering the
// groups by hash instead of by first path fails this test.
recs := []scanRec{ recs := []scanRec{
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/2"}, {size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"},
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/1"}, {size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"},
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/2"}, {size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"},
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/1"}, {size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"},
} }
groups := dupeGroupsOf(t, recs) groups := collectDupeGroups(recs)
if len(groups) != 2 { if len(groups) != 2 {
t.Fatalf("len(groups) = %d, want 2", len(groups)) t.Fatalf("len(groups) = %d, want 2", len(groups))
} }
@@ -347,7 +218,7 @@ func TestDupeGroupsTieBreak(t *testing.T) {
} }
} }
func TestDupeGroupsDeterministic(t *testing.T) { func TestCollectDupeGroupsDeterministic(t *testing.T) {
t.Parallel() t.Parallel()
recs := []scanRec{ recs := []scanRec{
@@ -357,12 +228,12 @@ func TestDupeGroupsDeterministic(t *testing.T) {
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"}, {size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
} }
forward := dupeGroupsOf(t, recs) forward := collectDupeGroups(recs)
reversed := slices.Clone(recs) reversed := slices.Clone(recs)
slices.Reverse(reversed) slices.Reverse(reversed)
backward := dupeGroupsOf(t, reversed) backward := collectDupeGroups(reversed)
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool { if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
return a.size == b.size && slices.Equal(a.paths, b.paths) return a.size == b.size && slices.Equal(a.paths, b.paths)
}) { }) {
+26 -98
View File
@@ -11,7 +11,6 @@ import (
"io" "io"
"io/fs" "io/fs"
"os" "os"
"os/signal"
"path/filepath" "path/filepath"
"slices" "slices"
"strings" "strings"
@@ -58,10 +57,6 @@ const sampleWindow = 1024 * 1024
// and hash worker pools. // and hash worker pools.
const workQueueDepth = 1024 const workQueueDepth = 1024
// errInterrupted reports a scan stopped by SIGINT or SIGTERM. runScan
// has already printed its line, so run prints nothing more.
var errInterrupted = errors.New("scan interrupted")
// fileRec carries one statted file between the scan phases. dev and // fileRec carries one statted file between the scan phases. dev and
// ino identify the underlying inode so hard-linked paths can share // ino identify the underlying inode so hard-linked paths can share
// one read; both are zero when the platform exposes no inode. // one read; both are zero when the platform exposes no inode.
@@ -95,14 +90,15 @@ type fileMeta struct {
// scan fails before it walks the filesystem or opens the database. // scan fails before it walks the filesystem or opens the database.
// Errors are returned rather than exiting, so that the deferred close — // Errors are returned rather than exiting, so that the deferred close —
// which takes the database out of WAL mode — always runs, and the lock // which takes the database out of WAL mode — always runs, and the lock
// is released after it. When ctx is cancelled, as by the SIGINT or // is released after it. Cancelling ctx unwinds the worker pools and
// SIGTERM that interruptContext catches, the scan keeps what it has // aborts the scan with the context's error.
// hashed (see syncScan), prints how many files its walk reached, and
// returns errInterrupted. workers must be at least 1; the scan command
// rejects anything less.
func runScan(ctx context.Context, roots []string, workers int, func runScan(ctx context.Context, roots []string, workers int,
oneFS bool, oneFS bool,
) error { ) error {
if workers < 1 {
workers = 1
}
roots, err := resolveRoots(roots) roots, err := resolveRoots(roots)
if err != nil { if err != nil {
return err return err
@@ -118,12 +114,6 @@ func runScan(ctx context.Context, roots []string, workers int,
defer func() { _ = lock.Close() }() defer func() { _ = lock.Close() }()
db, err := openScanDatabase(ctx, dbPath) db, err := openScanDatabase(ctx, dbPath)
if err != nil && ctx.Err() != nil {
// Interrupted while opening; SQLite may report that with an
// error of its own rather than the context's.
return interrupted(0)
}
if err != nil { if err != nil {
return err return err
} }
@@ -131,10 +121,6 @@ func runScan(ctx context.Context, roots []string, workers int,
defer closeScanDatabase(ctx, db, dbPath) defer closeScanDatabase(ctx, db, dbPath)
st, err := syncScan(ctx, db, roots, workers, oneFS) st, err := syncScan(ctx, db, roots, workers, oneFS)
if errors.Is(err, context.Canceled) {
return interrupted(st.walked)
}
if err != nil { if err != nil {
return fmt.Errorf("update database %s: %w", dbPath, err) return fmt.Errorf("update database %s: %w", dbPath, err)
} }
@@ -148,34 +134,6 @@ func runScan(ctx context.Context, roots []string, workers int,
return nil return nil
} }
// interruptContext returns a copy of ctx that the first SIGINT or
// SIGTERM cancels; the scan command runs the scan under it. stop
// releases the signals.
func interruptContext(ctx context.Context) (context.Context, func()) {
// A SIGINT ignored from the start, as by a script's background job,
// stays ignored.
signals := []os.Signal{syscall.SIGTERM}
if !signal.Ignored(syscall.SIGINT) {
signals = append(signals, syscall.SIGINT)
}
ctx, stop := signal.NotifyContext(ctx, signals...)
// Stopping restores the default handling, so a second signal ends
// the process at once.
context.AfterFunc(ctx, stop)
return ctx, stop
}
// interrupted prints the line for a scan stopped by a signal after its
// walk reached walked files, and returns errInterrupted.
func interrupted(walked int) error {
fmt.Fprintf(os.Stderr, "scan: interrupted after %d files\n", walked)
return errInterrupted
}
// resolveRoots converts each PATH operand to an absolute, lexically // resolveRoots converts each PATH operand to an absolute, lexically
// cleaned path (symlinks are not resolved) and verifies that it // cleaned path (symlinks are not resolved) and verifies that it
// exists. Database records are keyed by absolute path, so scan results // exists. Database records are keyed by absolute path, so scan results
@@ -229,9 +187,8 @@ func pruneRoots(roots []string) []string {
} }
// scanStats summarizes one scan's database synchronization for the // scanStats summarizes one scan's database synchronization for the
// final stderr summary, or for the line an interrupted scan prints. // final stderr summary.
type scanStats struct { type scanStats struct {
walked int // files the walk reached
added int added int
updated int updated int
removed int removed int
@@ -253,30 +210,7 @@ type scanState struct {
st scanStats st scanStats
} }
// syncScan synchronizes the database with the filesystem under roots; // syncScan synchronizes the database with the filesystem under roots
// see runPhases. When ctx is cancelled, as by an interrupt, it commits
// the hashed records still waiting in the batch, starts no other write
// or deletion, and returns the cancellation.
func syncScan(ctx context.Context, db *sql.DB, roots []string,
workers int, oneFS bool,
) (scanStats, error) {
s := &scanState{db: db}
err := s.runPhases(ctx, roots, workers, oneFS)
if err == nil || ctx.Err() == nil {
return s.st, err
}
// The one write made after the cancellation, so it cannot use ctx.
err = applyChanges(context.WithoutCancel(ctx), db, s.batch, nil, nil)
if err != nil {
return s.st, err
}
return s.st, ctx.Err()
}
// runPhases synchronizes the database with the filesystem under roots
// in four sequential phases: walk (enumerate and stat every file, // in four sequential phases: walk (enumerate and stat every file,
// building a complete size census), hash (read only the new or // building a complete size census), hash (read only the new or
// changed — or previously unhashed — files whose size at least one // changed — or previously unhashed — files whose size at least one
@@ -289,41 +223,48 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string,
// their content hash. Operands the walk cannot start from are dropped // their content hash. Operands the walk cannot start from are dropped
// first, so the records beneath them count as outside the roots unless // first, so the records beneath them count as outside the roots unless
// they lie under another root. // they lie under another root.
func (s *scanState) runPhases(ctx context.Context, roots []string, func syncScan(ctx context.Context, db *sql.DB, roots []string,
workers int, oneFS bool, workers int, oneFS bool,
) error { ) (scanStats, error) {
s := &scanState{db: db}
// Types are checked before pruning so that an operand under a // Types are checked before pruning so that an operand under a
// dropped one is still scanned, not dropped as lying under it. // dropped one is still scanned, not dropped as lying under it.
roots = pruneRoots(s.walkableRoots(roots)) roots = pruneRoots(s.walkableRoots(roots))
err := s.loadIndex(ctx, roots) err := s.loadIndex(ctx, roots)
if err != nil { if err != nil {
return err return s.st, err
} }
changed, unhashed := s.walkPhase(startWalk(ctx, roots, oneFS, workers)) changed, unhashed := s.walkPhase(startWalk(ctx, roots, oneFS, workers))
// A cancelled walk stops early, so its size census covers only part // A cancelled walk stops early, so its size census covers only part
// of the roots, and every file it never reached would look vanished // of the roots, and every file it never reached looks vanished to
// to the update phase. Stop before anything is written or deleted. // the update phase. Defence in depth rather than the only barrier:
// that phase would today fail on its first BeginTx with the same
// cancelled context before deleting anything. But it is the barrier
// that survives a later decision to let an interrupted scan commit
// what it has, and it turns a confusing failure deep in the update
// phase into a clean abort at the phase boundary.
err = ctx.Err() err = ctx.Err()
if err != nil { if err != nil {
return err return s.st, err
} }
s.partition(changed, unhashed) s.partition(changed, unhashed)
err = s.hashPhase(ctx, workers) err = s.hashPhase(ctx, workers)
if err != nil { if err != nil {
return err return s.st, err
} }
err = s.updatePhase(ctx) err = s.updatePhase(ctx)
if err != nil { if err != nil {
return err return s.st, err
} }
return s.contentPhase(ctx, workers) return s.st, s.contentPhase(ctx, workers)
} }
// walkableRoots returns the operands the walk can start from: regular // walkableRoots returns the operands the walk can start from: regular
@@ -408,7 +349,6 @@ func (s *scanState) walkPhase(
} }
s.sizes = append(s.sizes, ev.rec.size) s.sizes = append(s.sizes, ev.rec.size)
s.st.walked++
prog.increment() prog.increment()
@@ -612,22 +552,16 @@ func (s *scanState) recordRun(ctx context.Context, r hashResult) error {
} }
// commitFullBatch commits the running batch once it holds // commitFullBatch commits the running batch once it holds
// updateBatchSize records. A batch that fails to commit is kept: the // updateBatchSize records.
// commit fails when the scan is interrupted, and syncScan then commits
// the batch itself.
func (s *scanState) commitFullBatch(ctx context.Context) error { func (s *scanState) commitFullBatch(ctx context.Context) error {
if len(s.batch) < updateBatchSize { if len(s.batch) < updateBatchSize {
return nil return nil
} }
err := applyBatch(ctx, s.db, s.batch, nil, nil) err := applyBatch(ctx, s.db, s.batch, nil, nil)
if err != nil {
return err
}
s.batch = s.batch[:0] s.batch = s.batch[:0]
return nil return err
} }
// updatePhase writes the scan's tail under one progress display: the // updatePhase writes the scan's tail under one progress display: the
@@ -1070,12 +1004,6 @@ func walkOneDir(ctx context.Context, job dirJob, oneFS bool,
var subs []dirJob var subs []dirJob
for _, e := range entries { for _, e := range entries {
// A cancelled scan wants nothing more from this directory: stop
// rather than lstat the rest of a large one.
if ctx.Err() != nil {
return nil
}
p := filepath.Join(job.path, e.Name()) p := filepath.Join(job.path, e.Name())
if e.IsDir() { if e.IsDir() {
+40 -205
View File
@@ -6,18 +6,14 @@ import (
"crypto/sha256" "crypto/sha256"
"database/sql" "database/sql"
"encoding/hex" "encoding/hex"
"errors"
"fmt" "fmt"
"io" "io"
"io/fs"
"os" "os"
"os/signal"
"path/filepath" "path/filepath"
"runtime" "runtime"
"slices" "slices"
"strconv" "strconv"
"strings" "strings"
"syscall"
"testing" "testing"
"time" "time"
) )
@@ -404,7 +400,7 @@ func TestScanContentGate(t *testing.T) {
} }
} }
groups := dupeGroups(t, db) groups := collectDupeGroups(recs)
if len(groups) != 1 || !slices.Equal(groups[0].paths, same) { if len(groups) != 1 || !slices.Equal(groups[0].paths, same) {
t.Fatalf("groups = %+v, want only the identical pair %q", t.Fatalf("groups = %+v, want only the identical pair %q",
groups, same) groups, same)
@@ -439,7 +435,7 @@ func TestScanContentAcrossOperands(t *testing.T) {
want := []string{a, b} want := []string{a, b}
slices.Sort(want) slices.Sort(want)
groups := dupeGroups(t, db) groups := collectDupeGroups(recs)
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) { if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
t.Fatalf("groups = %+v, want the pair %q", groups, want) t.Fatalf("groups = %+v, want the pair %q", groups, want)
} }
@@ -460,13 +456,13 @@ func TestScanContentWithinOperand(t *testing.T) {
added := sparseFile(t, dir, "d2", headTailMin) added := sparseFile(t, dir, "d2", headTailMin)
st := syncTree(t, db, dir) st := syncTree(t, db, dir)
if st != (scanStats{walked: 3, added: 1, unchanged: 2}) { if st != (scanStats{added: 1, unchanged: 2}) {
t.Fatalf("rescan stats = %+v, want 1 added 2 unchanged", st) t.Fatalf("rescan stats = %+v, want 1 added 2 unchanged", st)
} }
want := []string{stored, added} want := []string{stored, added}
groups := dupeGroups(t, db) groups := collectDupeGroups(dbRecords(t, db))
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) { if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
t.Fatalf("groups = %+v, want the pair %q", groups, want) t.Fatalf("groups = %+v, want the pair %q", groups, want)
} }
@@ -506,7 +502,7 @@ func TestScanContentStalePartners(t *testing.T) {
sparseFile(t, dirB, "changed-copy", headTailMin+1) sparseFile(t, dirB, "changed-copy", headTailMin+1)
st := syncTree(t, db, dirB) st := syncTree(t, db, dirB)
if st != (scanStats{walked: 2, added: 2}) { if st != (scanStats{added: 2}) {
t.Errorf("stats = %+v, want 2 added and nothing skipped", st) t.Errorf("stats = %+v, want 2 added and nothing skipped", st)
} }
@@ -524,7 +520,7 @@ func TestScanContentStalePartners(t *testing.T) {
} }
} }
if groups := dupeGroups(t, db); len(groups) != 0 { if groups := collectDupeGroups(recs); len(groups) != 0 {
t.Errorf("groups = %+v, want none", groups) t.Errorf("groups = %+v, want none", groups)
} }
} }
@@ -563,7 +559,7 @@ func TestScanContentHashedStalePartners(t *testing.T) {
b := sparseFile(t, t.TempDir(), "copy", headTailMin) b := sparseFile(t, t.TempDir(), "copy", headTailMin)
st := syncTree(t, db, filepath.Dir(b)) st := syncTree(t, db, filepath.Dir(b))
if st != (scanStats{walked: 1, added: 1}) { if st != (scanStats{added: 1}) {
t.Errorf("stats = %+v, want 1 added and nothing skipped", st) t.Errorf("stats = %+v, want 1 added and nothing skipped", st)
} }
@@ -575,7 +571,7 @@ func TestScanContentHashedStalePartners(t *testing.T) {
// The stored records lie outside the operand and are left as they // The stored records lie outside the operand and are left as they
// are, so they still group with each other, but not with the copy. // are, so they still group with each other, but not with the copy.
groups := dupeGroups(t, db) groups := collectDupeGroups(recs)
if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) { if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) {
t.Errorf("groups = %+v, want only the stored pair %q", groups, stored) t.Errorf("groups = %+v, want only the stored pair %q", groups, stored)
} }
@@ -603,7 +599,7 @@ func TestScanContentReadFailure(t *testing.T) {
b := sparseFile(t, dirB, "b", headTailMin) b := sparseFile(t, dirB, "b", headTailMin)
st := syncTree(t, db, dirB) st := syncTree(t, db, dirB)
if st != (scanStats{walked: 1, added: 1, skipped: 1}) { if st != (scanStats{added: 1, skipped: 1}) {
t.Fatalf("stats = %+v, want 1 added 1 skipped", st) t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
} }
@@ -617,14 +613,14 @@ func TestScanContentReadFailure(t *testing.T) {
} }
st = syncTree(t, db, dirB) st = syncTree(t, db, dirB)
if st != (scanStats{walked: 1, unchanged: 1}) { if st != (scanStats{unchanged: 1}) {
t.Fatalf("rescan stats = %+v, want 1 unchanged", st) t.Fatalf("rescan stats = %+v, want 1 unchanged", st)
} }
want := []string{a, b} want := []string{a, b}
slices.Sort(want) slices.Sort(want)
groups := dupeGroups(t, db) groups := collectDupeGroups(dbRecords(t, db))
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) { if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
t.Fatalf("groups = %+v, want the pair %q after the retry", t.Fatalf("groups = %+v, want the pair %q after the retry",
groups, want) groups, want)
@@ -663,7 +659,7 @@ func TestScanContentCheckError(t *testing.T) {
b := sparseFile(t, t.TempDir(), "b", headTailMin) b := sparseFile(t, t.TempDir(), "b", headTailMin)
st := syncTree(t, db, filepath.Dir(b)) st := syncTree(t, db, filepath.Dir(b))
if st != (scanStats{walked: 1, added: 1, skipped: 1}) { if st != (scanStats{added: 1, skipped: 1}) {
t.Fatalf("stats = %+v, want 1 added 1 skipped", st) t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
} }
@@ -691,7 +687,7 @@ func TestScanContentHardlinks(t *testing.T) {
c := sparseFile(t, dir, "copy", headTailMin) c := sparseFile(t, dir, "copy", headTailMin)
st := syncTree(t, db, dir) st := syncTree(t, db, dir)
if st != (scanStats{walked: 3, added: 3}) { if st != (scanStats{added: 3}) {
t.Fatalf("stats = %+v, want 3 added", st) t.Fatalf("stats = %+v, want 3 added", st)
} }
@@ -914,159 +910,6 @@ func TestDeviceOfInfo(t *testing.T) {
} }
} }
// dirEntryFor returns the fs.DirEntry for name within dir, obtained via
// the same os.ReadDir the walk uses, so it carries a real Info().
func dirEntryFor(t *testing.T, dir, name string) fs.DirEntry {
t.Helper()
entries, err := os.ReadDir(dir)
if err != nil {
t.Fatal(err)
}
for _, e := range entries {
if e.Name() == name {
return e
}
}
t.Fatalf("entry %q not found in %q", name, dir)
return nil
}
// callSubdirJob runs subdirJob against e under parent, collecting any
// warning events it emits (subdirJob emits at most one).
func callSubdirJob(t *testing.T, p string, e fs.DirEntry,
parent dirJob, oneFS bool,
) (dirJob, bool, []walkEvent) {
t.Helper()
events := make(chan walkEvent, 1)
job, ok := subdirJob(t.Context(), p, e, parent, oneFS, events)
close(events)
var evs []walkEvent
for ev := range events {
evs = append(evs, ev)
}
return job, ok, evs
}
// TestSubdirJobOneFilesystem exercises the -x boundary check in
// subdirJob directly, so no second real filesystem is needed. The
// subdirectory's real device is compared against a fabricated operand
// device.
func TestSubdirJobOneFilesystem(t *testing.T) {
t.Parallel()
dir := t.TempDir()
sub := filepath.Join(dir, "sub")
err := os.Mkdir(sub, 0o750)
if err != nil {
t.Fatal(err)
}
info, err := os.Lstat(sub)
if err != nil {
t.Fatal(err)
}
dev, ok := deviceOfInfo(info)
if !ok {
t.Skip("platform exposes no device id")
}
// A device the subdirectory is not on, standing in for an operand
// rooted on a different filesystem.
otherDev := dev + 1
e := dirEntryFor(t, dir, "sub")
cases := []struct {
name string
oneFS bool
parent dirJob
wantOK bool
}{
// -x on, subdirectory on a different device than its operand:
// descent is refused.
{"reject across boundary", true,
dirJob{rootDev: otherDev, rootDevOK: true}, false},
// -x on but the operand's own device is unknown: the boundary
// check is bypassed and descent proceeds.
{"bypass when root device unknown", true,
dirJob{rootDev: otherDev, rootDevOK: false}, true},
// Default (no -x): boundaries are crossed even onto a different
// device.
{"cross by default", false,
dirJob{rootDev: otherDev, rootDevOK: true}, true},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
t.Parallel()
job, ok, evs := callSubdirJob(t, sub, e, tc.parent, tc.oneFS)
if ok != tc.wantOK {
t.Fatalf("accepted = %v, want %v", ok, tc.wantOK)
}
if len(evs) != 0 {
t.Fatalf("unexpected events: %+v", evs)
}
// The accepted job must carry the operand's device down, or -x
// stops checking below the first level.
want := dirJob{
path: sub,
rootDev: tc.parent.rootDev,
rootDevOK: tc.parent.rootDevOK,
}
if ok && job != want {
t.Fatalf("job = %+v, want %+v", job, want)
}
})
}
}
// errInfoUnavailable is returned by errDirEntry.Info().
var errInfoUnavailable = errors.New("info unavailable")
// errDirEntry is a directory entry whose Info() always fails, driving
// subdirJob's stat-error branch deterministically.
type errDirEntry struct{ name string }
func (e errDirEntry) Name() string { return e.name }
func (errDirEntry) IsDir() bool { return true }
func (errDirEntry) Type() fs.FileMode {
return fs.ModeDir
}
func (errDirEntry) Info() (fs.FileInfo, error) {
return nil, errInfoUnavailable
}
// TestSubdirJobStatError asserts that when a subdirectory's Info()
// fails under -x, subdirJob warns and refuses descent.
func TestSubdirJobStatError(t *testing.T) {
t.Parallel()
p := "/does/not/matter/sub"
_, ok, evs := callSubdirJob(t, p, errDirEntry{name: "sub"},
dirJob{rootDev: 1, rootDevOK: true}, true)
if ok {
t.Fatal("descent accepted after stat error, want refused")
}
if len(evs) != 1 || !evs[0].fail || !strings.Contains(evs[0].warn, p) {
t.Fatalf("want one warning naming the path, got %+v", evs)
}
}
// buildSmokeTree recreates the README smoke-test filesystem layout // buildSmokeTree recreates the README smoke-test filesystem layout
// with deterministic content and returns the tree root. // with deterministic content and returns the tree root.
func buildSmokeTree(t *testing.T) string { func buildSmokeTree(t *testing.T) string {
@@ -1113,16 +956,11 @@ func syncTree(t *testing.T, db *sql.DB, roots ...string) scanStats {
return st return st
} }
// dbRecords returns every record currently in the database, in path // dbRecords returns every record currently in the database.
// order.
func dbRecords(t *testing.T, db *sql.DB) []scanRec { func dbRecords(t *testing.T, db *sql.DB) []scanRec {
t.Helper() t.Helper()
var recs []scanRec recs, err := loadFileRows(t.Context(), db)
err := loadFileRows(t.Context(), db, func(r scanRec) {
recs = append(recs, r)
})
if err != nil { if err != nil {
t.Fatal(err) t.Fatal(err)
} }
@@ -1159,10 +997,10 @@ func recordPaths(recs []scanRec) []string {
// assertSmokeDupeGroups checks the file-level duplicate groups for the // assertSmokeDupeGroups checks the file-level duplicate groups for the
// smoke tree rooted at dir. // smoke tree rooted at dir.
func assertSmokeDupeGroups(t *testing.T, dir string, db *sql.DB) { func assertSmokeDupeGroups(t *testing.T, dir string, parsed []scanRec) {
t.Helper() t.Helper()
groups := dupeGroups(t, db) groups := collectDupeGroups(parsed)
if len(groups) != 5 { if len(groups) != 5 {
t.Fatalf("len(groups) = %d, want 5", len(groups)) t.Fatalf("len(groups) = %d, want 5", len(groups))
} }
@@ -1187,10 +1025,11 @@ func assertSmokeDupeGroups(t *testing.T, dir string, db *sql.DB) {
// assertSmokeTreeGroups checks the duplicate-tree groups for the smoke // assertSmokeTreeGroups checks the duplicate-tree groups for the smoke
// tree rooted at dir. // tree rooted at dir.
func assertSmokeTreeGroups(t *testing.T, dir string, db *sql.DB) { func assertSmokeTreeGroups(t *testing.T, dir string, parsed []scanRec) {
t.Helper() t.Helper()
super, dirs := dbTree(t, db) super, dirs := buildHierarchy(parsed)
super.compute()
tg := collectTreeGroups(dirs, super) tg := collectTreeGroups(dirs, super)
if len(tg) != 1 { if len(tg) != 1 {
@@ -1215,7 +1054,7 @@ func TestScanPipeline(t *testing.T) {
db := openTestDB(t) db := openTestDB(t)
st := syncTree(t, db, dir) st := syncTree(t, db, dir)
if st != (scanStats{walked: smokeTreeFiles, added: smokeTreeFiles}) { if st != (scanStats{added: smokeTreeFiles}) {
t.Fatalf("stats = %+v, want %d added only", st, smokeTreeFiles) t.Fatalf("stats = %+v, want %d added only", st, smokeTreeFiles)
} }
@@ -1224,8 +1063,8 @@ func TestScanPipeline(t *testing.T) {
t.Fatalf("len(records) = %d, want %d", len(parsed), smokeTreeFiles) t.Fatalf("len(records) = %d, want %d", len(parsed), smokeTreeFiles)
} }
assertSmokeDupeGroups(t, dir, db) assertSmokeDupeGroups(t, dir, parsed)
assertSmokeTreeGroups(t, dir, db) assertSmokeTreeGroups(t, dir, parsed)
} }
func TestSyncScanUnchangedReuse(t *testing.T) { func TestSyncScanUnchangedReuse(t *testing.T) {
@@ -1238,7 +1077,7 @@ func TestSyncScanUnchangedReuse(t *testing.T) {
writeFile(t, dir, "b.bin", pattern(2, 600)) writeFile(t, dir, "b.bin", pattern(2, 600))
st := syncTree(t, db, dir) st := syncTree(t, db, dir)
if st != (scanStats{walked: 2, added: 2}) { if st != (scanStats{added: 2}) {
t.Fatalf("first scan stats = %+v, want 2 added", st) t.Fatalf("first scan stats = %+v, want 2 added", st)
} }
@@ -1252,7 +1091,7 @@ func TestSyncScanUnchangedReuse(t *testing.T) {
} }
st = syncTree(t, db, dir) st = syncTree(t, db, dir)
if st != (scanStats{walked: 2, unchanged: 2}) { if st != (scanStats{unchanged: 2}) {
t.Fatalf("rescan stats = %+v, want 2 unchanged", st) t.Fatalf("rescan stats = %+v, want 2 unchanged", st)
} }
@@ -1281,7 +1120,7 @@ func TestSyncScanMtimeBump(t *testing.T) {
} }
st := syncTree(t, db, dir) st := syncTree(t, db, dir)
if st != (scanStats{walked: 1, updated: 1}) { if st != (scanStats{updated: 1}) {
t.Fatalf("mtime-bump stats = %+v, want 1 updated", st) t.Fatalf("mtime-bump stats = %+v, want 1 updated", st)
} }
@@ -1309,7 +1148,7 @@ func TestSyncScanAddRemove(t *testing.T) {
} }
st := syncTree(t, db, dir) st := syncTree(t, db, dir)
if st != (scanStats{walked: 2, added: 1, removed: 1, unchanged: 1}) { if st != (scanStats{added: 1, removed: 1, unchanged: 1}) {
t.Fatalf("add/remove stats = %+v, want 1 added 1 removed 1 unchanged", t.Fatalf("add/remove stats = %+v, want 1 added 1 removed 1 unchanged",
st) st)
} }
@@ -1426,7 +1265,7 @@ func TestSyncScanOverlappingRoots(t *testing.T) {
// A file reachable via two overlapping operands is deduplicated // A file reachable via two overlapping operands is deduplicated
// by path in the shared walk and processed once. // by path in the shared walk and processed once.
st := syncTree(t, db, dir, filepath.Join(dir, "sub")) st := syncTree(t, db, dir, filepath.Join(dir, "sub"))
if st != (scanStats{walked: 1, added: 1}) { if st != (scanStats{added: 1}) {
t.Fatalf("stats = %+v, want 1 added", st) t.Fatalf("stats = %+v, want 1 added", st)
} }
@@ -1447,7 +1286,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
// Neither size is shared, so neither file is read: both records // Neither size is shared, so neither file is read: both records
// are written without hashes and no duplicates are reported. // are written without hashes and no duplicates are reported.
st := syncTree(t, db, dir) st := syncTree(t, db, dir)
if st != (scanStats{walked: 2, added: 2}) { if st != (scanStats{added: 2}) {
t.Fatalf("stats = %+v, want 2 added", st) t.Fatalf("stats = %+v, want 2 added", st)
} }
@@ -1459,7 +1298,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
} }
} }
if groups := dupeGroups(t, db); len(groups) != 0 { if groups := collectDupeGroups(recs); len(groups) != 0 {
t.Fatalf("groups = %+v, want none from unhashed records", groups) t.Fatalf("groups = %+v, want none from unhashed records", groups)
} }
@@ -1469,12 +1308,12 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
c := writeFile(t, dir, "c.bin", pattern(1, 500)) c := writeFile(t, dir, "c.bin", pattern(1, 500))
st = syncTree(t, db, dir) st = syncTree(t, db, dir)
if st != (scanStats{walked: 3, added: 1, updated: 1, unchanged: 1}) { if st != (scanStats{added: 1, updated: 1, unchanged: 1}) {
t.Fatalf("rescan stats = %+v, want 1 added 1 updated 1 unchanged", t.Fatalf("rescan stats = %+v, want 1 added 1 updated 1 unchanged",
st) st)
} }
groups := dupeGroups(t, db) groups := collectDupeGroups(dbRecords(t, db))
if len(groups) != 1 { if len(groups) != 1 {
t.Fatalf("groups = %+v, want the a/c pair", groups) t.Fatalf("groups = %+v, want the a/c pair", groups)
} }
@@ -1510,7 +1349,8 @@ func TestTreesUnhashedNeverEqual(t *testing.T) {
} }
for name, unknown := range cases { for name, unknown := range cases {
super, dirs := treeOf(t, append(slices.Clone(shared), unknown...)) super, dirs := buildHierarchy(append(slices.Clone(shared), unknown...))
super.compute()
if tg := collectTreeGroups(dirs, super); len(tg) != 0 { if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
t.Errorf("%s: tree groups = %d, want 0 (the files may differ)", t.Errorf("%s: tree groups = %d, want 0 (the files may differ)",
@@ -1533,7 +1373,7 @@ func TestScanHardlinksReadOnce(t *testing.T) {
} }
st := syncTree(t, db, dir) st := syncTree(t, db, dir)
if st != (scanStats{walked: 2, added: 2}) { if st != (scanStats{added: 2}) {
t.Fatalf("stats = %+v, want 2 added", st) t.Fatalf("stats = %+v, want 2 added", st)
} }
@@ -1547,7 +1387,7 @@ func TestScanHardlinksReadOnce(t *testing.T) {
t.Fatalf("hardlink hashes differ: %+v vs %+v", ra, rb) t.Fatalf("hardlink hashes differ: %+v vs %+v", ra, rb)
} }
if groups := dupeGroups(t, db); len(groups) != 1 { if groups := collectDupeGroups(recs); len(groups) != 1 {
t.Fatalf("groups = %+v, want the hardlink pair", groups) t.Fatalf("groups = %+v, want the hardlink pair", groups)
} }
} }
@@ -1655,13 +1495,6 @@ func injectWriteFailure(t *testing.T, path string) {
func baselineGoroutines(t *testing.T) int { func baselineGoroutines(t *testing.T) int {
t.Helper() t.Helper()
// The first scan command in a process starts os/signal's goroutine,
// which never exits. Start it now, so the baseline counts it
// instead of the scan seeming to leave it behind.
ch := make(chan os.Signal, 1)
signal.Notify(ch, syscall.SIGINT)
signal.Stop(ch)
deadline := time.Now().Add(goroutineSettle) deadline := time.Now().Add(goroutineSettle)
last := runtime.NumGoroutine() last := runtime.NumGoroutine()
@@ -1795,8 +1628,10 @@ func TestReportsNeverTouchFilesystem(t *testing.T) {
t.Fatal(err) t.Fatal(err)
} }
assertSmokeDupeGroups(t, dir, db) recs := dbRecords(t, db)
assertSmokeTreeGroups(t, dir, db)
assertSmokeDupeGroups(t, dir, recs)
assertSmokeTreeGroups(t, dir, recs)
} }
func TestUnderRoot(t *testing.T) { func TestUnderRoot(t *testing.T) {
+21 -12
View File
@@ -1,9 +1,12 @@
#!/bin/sh #!/bin/sh
# script/bootstrap: install all dependencies needed to build and develop # script/bootstrap: install all dependencies needed to build and develop
# this repo. Idempotent; assumes nothing is present (not git, make, or # this repo. Idempotent: every install is guarded by a check so already
# go). Base tooling comes from nix, apt, brew, or apk (detected in that # installed tools are skipped. Base tooling comes from nix, apt, brew,
# order). golangci-lint and prettier are never installed: they run via # or apk (detected in that order); assumes nothing is present (not git,
# docker only (script/lint, script/fmt, script/fmt-check). # make, or go). The linter is NOT installed: golangci-lint runs via
# docker only (script/lint), pinned by image digest, so the only lint
# prerequisite is a working docker — which is warned about, not
# installed, because everything except linting works without it.
set -eu set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)" ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
@@ -58,19 +61,25 @@ missing() {
main() { main() {
cd "$ROOT" cd "$ROOT"
# Deliberately unpinned, so presence is the whole check: go.mod # System tooling, deliberately unpinned: these come from the host
# governs the Go version, and reproducible builds run in the # package manager and whatever version it ships is what the host
# digest-pinned Docker images. # gets, so a presence check is the right check. The repo pins no
# system toolchain versions — the Go language version is governed by
# go.mod, and builds that must be reproducible run in the Docker
# image, whose base images are pinned by digest.
if missing git; then pkg_install git git git git; fi if missing git; then pkg_install git git git git; fi
if missing make; then pkg_install gnumake make make make; fi if missing make; then pkg_install gnumake make make make; fi
if missing go; then pkg_install go golang go go; fi if missing go; then pkg_install go golang go go; fi
# Warn, do not fail: only the targets named below, and the # Linting runs via docker only (script/lint), so docker is a lint
# pre-commit hook, need docker. # prerequisite rather than something bootstrap installs. Warn, do
# not fail: everything except `make lint` — and, through it,
# `make check`, `make docker` and the pre-commit hook — works
# without it.
if missing docker; then if missing docker; then
echo "bootstrap: WARNING: docker not found; make lint, make fmt," >&2 echo "bootstrap: WARNING: docker not found; make lint, make check" >&2
echo "bootstrap: make fmt-check, make check, make docker and" >&2 echo "bootstrap: and make docker require it. Install docker to" >&2
echo "bootstrap: make test-race require it." >&2 echo "bootstrap: run the linter." >&2
fi fi
go mod download go mod download
+22 -7
View File
@@ -1,19 +1,34 @@
#!/bin/sh #!/bin/sh
# script/cibuild: run the CI build. The Gitea workflow runs this on # script/cibuild: run the CI build. The Gitea workflow runs this on
# push. The Dockerfile runs every gate make check runs, as build steps, # push.
# so a successful build means the repo is green.
# #
# Without a fresh CHECK_EPOCH, a rebuild of an unchanged checkout serves # The Dockerfile runs the gates individually as build steps, not the
# the gate layers from cache and passes having run none of them. The # make check aggregate: the lint stage runs make fmt-check,
# process id goes in with the epoch so two runs started in the same # script/verify-lint-image-pin, golangci-lint config verify and
# second still differ. # golangci-lint run; the build stage, dropped to an unprivileged user,
# runs make test and make fmt-check. Neither make lint nor make check
# appears, because both reach script/lint, which is itself a docker
# build, and a docker build cannot run inside one. Lint is not skipped
# by that — the linter is invoked directly in the lint stage, and the
# build stage's COPY --from=lint makes that stage a prerequisite, so
# BuildKit must finish it first. Between the two stages everything
# make check would run has run, which is why a successful build here
# implies the repo is green.
#
# That implication holds only because of CHECK_EPOCH. A COPY layer is
# invalidated only by changed content, and a rebuild of an unchanged
# checkout sends the same content, so without a fresh value here Docker
# serves the gate layers from cache and the build reports a green it
# never earned. Passing the current epoch invalidates the gate
# layers on every run while leaving the pinned base images and
# go mod download cached; see the Dockerfile for the placement.
set -eu set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)" ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() { main() {
cd "$ROOT" cd "$ROOT"
docker build --build-arg CHECK_EPOCH="$(date +%s)-$$" . docker build --build-arg CHECK_EPOCH="$(date +%s)" .
} }
main "$@" main "$@"
+10 -5
View File
@@ -1,9 +1,14 @@
#!/bin/sh #!/bin/sh
# script/docker: build the Docker image tagged with the project name. # script/docker: build the Docker image tagged with the project name.
# The tag comes from script/projectname. CHECK_EPOCH is passed for the # The tag comes from script/projectname.
# same reason script/cibuild passes it: without a fresh value an #
# unchanged tree is served from cache and this exits 0 having run no # CHECK_EPOCH is passed for the same reason script/cibuild passes it:
# gate. # without it Docker serves the Dockerfile's gate layers from cache on an
# unchanged tree and this exits 0 having run neither the lint stage's
# gates nor the builder stage's test and fmt-check gates. This is the
# set of gates a developer or reviewer runs by hand, so a cached pass
# here is the most misleading result the repo can produce. Dependency
# layers sit above the ARG and stay cached.
set -eu set -eu
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)" SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
@@ -12,7 +17,7 @@ ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
main() { main() {
cd "$ROOT" cd "$ROOT"
docker build \ docker build \
--build-arg CHECK_EPOCH="$(date +%s)-$$" \ --build-arg CHECK_EPOCH="$(date +%s)" \
-t "$("$SCRIPT_DIR/projectname")" \ -t "$("$SCRIPT_DIR/projectname")" \
. .
} }
+2 -12
View File
@@ -1,22 +1,12 @@
#!/bin/sh #!/bin/sh
# script/fmt: format all files (writes): the Go sources with gofmt, the # script/fmt: format all files (writes).
# Markdown with prettier. prettier is never installed on the host: it
# runs from the Dockerfile's prettier stage with the repository mounted,
# as the calling user so the files it rewrites keep their owner. The tag
# makes each build replace the previous image instead of leaving another
# one behind.
set -eu set -eu
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)" ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
main() { main() {
cd "$ROOT" cd "$ROOT"
gofmt -s -w . gofmt -s -w .
image="$("$SCRIPT_DIR/projectname")-prettier"
docker build -q --target prettier -t "$image" . >/dev/null
docker run --rm --user "$(id -u):$(id -g)" -v "$ROOT:/src" "$image" \
prettier --write '**/*.md' --tab-width 4 --prose-wrap always
} }
main "$@" main "$@"
+4 -25
View File
@@ -1,39 +1,18 @@
#!/bin/sh #!/bin/sh
# script/fmt-check: check formatting (read-only). Same scope as # script/fmt-check: check formatting (read-only). Same scope as
# script/fmt, but fails instead of writing. gofmt and prettier both run # script/fmt, but fails instead of writing.
# every time and each reports its own failure, so the output says which
# one failed.
set -eu set -eu
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)" ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
main() { main() {
cd "$ROOT" cd "$ROOT"
status=0 files="$(gofmt -s -l .)"
# Under set -e a bare assignment would end the script when gofmt
# fails (a Go file it cannot parse), and prettier would never run.
if ! files="$(gofmt -s -l .)"; then
echo "gofmt: failed; see its errors above" >&2
status=1
fi
if [ -n "$files" ]; then if [ -n "$files" ]; then
echo "gofmt: files not formatted:" >&2 echo "gofmt: files not formatted:" >&2
echo "$files" >&2 echo "$files" >&2
status=1 exit 1
fi fi
# Same image as script/fmt; see there.
image="$("$SCRIPT_DIR/projectname")-prettier"
docker build -q --target prettier -t "$image" . >/dev/null
if ! docker run --rm -v "$ROOT:/src:ro" "$image" \
prettier --check '**/*.md' --tab-width 4 --prose-wrap always; then
echo "prettier: Markdown not formatted; run make fmt" >&2
status=1
fi
exit "$status"
} }
main "$@" main "$@"
+13 -12
View File
@@ -1,17 +1,19 @@
#!/bin/sh #!/bin/sh
# script/lint: run the linter. golangci-lint is never installed on a # script/lint: run the linter. golangci-lint is never installed on a
# host: this builds Dockerfile.lint, which copies the repo into the # host: it runs via docker only, one way, everywhere — this builds
# digest-pinned golangci-lint image and lints as a build step, so a # Dockerfile.lint, which COPYs the repo into the digest-pinned
# successful build is a clean lint. A cold cache needs the network to # golangci-lint image and lints as a build step, so a successful build
# pull the image and for `go mod download`; once warm this runs offline # is a clean lint. The only prerequisite is a working docker. The gate
# until go.mod or go.sum changes. # steps make no network calls of their own, but Dockerfile.lint runs
# `go mod download` above them, so a cold cache does reach the network
# (as does pulling the pinned image); that layer stays cached, and once
# it is warm this runs offline until go.mod or go.sum changes.
# #
# Without a fresh CHECK_EPOCH docker serves the gate layers from cache # CHECK_EPOCH is what makes the result mean anything. Without it docker
# on an unchanged tree and this exits 0 having run no linter. The PID is # serves the gate layers from cache on an unchanged tree and this exits
# in the value because two lint runs land inside the same second easily. # 0 in well under a second having run no linter. The PID is in the value
# # as well as the epoch because two lint runs land inside the same second
# The image is never used, so --output=type=cacheonly writes none; # easily, and `date +%s` alone would cache the second one.
# without it every run leaves an untagged image behind.
set -eu set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)" ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
@@ -20,7 +22,6 @@ main() {
cd "$ROOT" cd "$ROOT"
docker build \ docker build \
--build-arg CHECK_EPOCH="$(date +%s)-$$" \ --build-arg CHECK_EPOCH="$(date +%s)-$$" \
--output=type=cacheonly \
-f Dockerfile.lint \ -f Dockerfile.lint \
. .
} }
-40
View File
@@ -1,40 +0,0 @@
#!/bin/sh
# script/test-race: run the test suite under the race detector. Not part
# of script/check.
#
# The race detector needs cgo and a C compiler, which the host build
# never uses, so the tests run in a golang image that has gcc. The
# checkout is mounted read-only, so the docker daemon must be local. The
# container starts with empty caches every time: each run downloads the
# dependencies and compiles them with the detector, which needs the
# network and takes minutes.
#
# The tests run as the calling user, never as root: several of them make
# a file unreadable and expect reading it to fail, and root reads it
# anyway. When the caller is root they run as nobody, and then the
# checkout must be readable by other users. Neither user has a home
# directory in the image, so HOME is /tmp, where Go puts its build cache.
set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
# golang:1.25-trixie, 2026-10-04. Debian rather than the Alpine image the
# Dockerfile builds with, because this one includes gcc.
IMAGE="golang@sha256:2c4c60ef415fbfa5e90300722293bef36c5e63fae17570ce18f580af933dbd73"
main() {
user="$(id -u):$(id -g)"
if [ "$(id -u)" -eq 0 ]; then
user=65534:65534
fi
docker run --rm \
--user "$user" \
--env HOME=/tmp \
--env CGO_ENABLED=1 \
--volume "$ROOT:/src:ro" \
--workdir /src \
"$IMAGE" \
go test -race -timeout 60s ./...
}
main "$@"
+15 -8
View File
@@ -2,16 +2,23 @@
# script/verify-lint-image-pin: fail unless the golangci-lint image # script/verify-lint-image-pin: fail unless the golangci-lint image
# referenced by Dockerfile.lint and the one referenced by the main # referenced by Dockerfile.lint and the one referenced by the main
# Dockerfile's lint stage are the same image at the same digest. Our own # Dockerfile's lint stage are the same image at the same digest. Our own
# extension to scripts-to-rule-them-all, not one of its entrypoints; run # extension to scripts-to-rule-them-all, not one of its entrypoints.
# as a gate in both files. Nothing else keeps the two pins in sync, and
# a bump applied to one alone would lint the same tree against different
# rulesets, both green.
# #
# Do not hardcode the expected digest here: that is a third copy to keep # The linter version is pinned in two independent files. That is the
# in sync. # shape #42 turned into a build failure rather than tolerate: nothing
# else keeps the two in sync, and a bump applied to one file alone would
# leave `make lint` and the fail-fast lint stage of `make docker`
# linting the same tree against different rulesets, both green. This is
# the single guard that stops it, run as a gate in both files.
# #
# A reference that cannot be read is a hard failure, not a skip: two # It deliberately restates neither pin. A hardcoded expected digest here
# empty strings compare equal. # would be a third copy — one more thing to bump, and the same drift one
# file further out. It compares the two files to each other and knows
# nothing about which version is correct.
#
# A reference that cannot be read is a hard failure, not a skip: a
# comparison of two empty strings succeeds, which would turn this guard
# into exactly the unearned green it exists to prevent.
set -eu set -eu
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)" ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
+103 -151
View File
@@ -12,48 +12,39 @@ import (
"strings" "strings"
) )
// treeNode is one directory reconstructed from the record paths. // fileSig is a file's duplicate signature; mtime is excluded.
type fileSig struct {
size int64
head string
tail string
content string
}
// treeNode is one directory reconstructed from the scan stream.
type treeNode struct { type treeNode struct {
path string path string
parent *treeNode parent *treeNode
// entries holds the serialized child entries until the digest is dirs map[string]*treeNode
// computed from them, and is then dropped. files map[string]fileSig
entries []string
digest [sha256.Size]byte digest [sha256.Size]byte
fileCount int64 fileCount int64
totalSize int64 totalSize int64
} }
// runTrees implements the trees subcommand: it reads every record from // runTrees implements the trees subcommand: it reads every record from
// the database in path order, reconstructs the directory hierarchy from // the database, reconstructs the directory hierarchy from the record
// the record paths, computes a Merkle-style digest per directory, and // paths, computes a Merkle-style digest per directory, and prints
// prints maximal duplicate-tree groups as TSV on stdout. It never // maximal duplicate-tree groups as TSV on stdout. It never touches the
// touches the scanned filesystem; its only I/O is the database, stdout, // scanned filesystem; its only I/O is the database, stdout, and
// and stderr. Any database problem, including a missing database, is // stderr.
// fatal.
func runTrees(ctx context.Context, stdout io.Writer) error { func runTrees(ctx context.Context, stdout io.Writer) error {
dbPath := databasePath() recs, err := loadRecords(ctx)
db, err := openReportDatabase(ctx, dbPath)
if err != nil { if err != nil {
return err return err
} }
defer func() { _ = db.Close() }() super, allDirs := buildHierarchy(recs)
super.compute()
records := 0
tree := newTreeBuilder()
err = loadFileRows(ctx, db, func(r scanRec) {
records++
tree.add(r)
})
if err != nil {
return fmt.Errorf("database %s: %w", dbPath, err)
}
super, allDirs := tree.finish()
dupes := collectTreeGroups(allDirs, super) dupes := collectTreeGroups(allDirs, super)
@@ -91,128 +82,73 @@ func runTrees(ctx context.Context, stdout io.Writer) error {
fmt.Fprintf(os.Stderr, fmt.Fprintf(os.Stderr,
"trees: %d records read, %d duplicate tree groups, %d dupe trees, "+ "trees: %d records read, %d duplicate tree groups, %d dupe trees, "+
"%s reclaimable\n", "%s reclaimable\n",
records, len(dupes), dupeTrees, humanBytes(reclaimable)) len(recs), len(dupes), dupeTrees, humanBytes(reclaimable))
return nil return nil
} }
// treeBuilder reconstructs the directory hierarchy from records added // buildHierarchy reconstructs the directory hierarchy from the record
// in path order, under a synthetic super-root. Paths are split on "/"; // paths under a synthetic super-root. Paths are split on "/"; for
// for absolute paths the first component is empty, which becomes the // absolute paths the first component is empty, which becomes the
// top-level directory with path "/". In path order all the paths under // top-level node with path "/". It returns the super-root and every
// one directory come together, so a directory is complete once a path // directory node created.
// outside it is added: its digest is computed then and its entries are func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
// dropped. Only the directories holding the latest path keep entries.
type treeBuilder struct {
super *treeNode
// open lists the directories holding the latest path, outermost
// first, starting with the super-root; names[i] is open[i]'s name.
open []*treeNode
names []string
// dirs lists every completed directory.
dirs []*treeNode
}
func newTreeBuilder() *treeBuilder {
super := &treeNode{} super := &treeNode{}
return &treeBuilder{ var allDirs []*treeNode
super: super,
open: []*treeNode{super},
names: []string{""},
}
}
// add adds one record. Each record must come after the previous one in for _, r := range recs {
// path order (byte order); otherwise a completed directory would be comps := strings.Split(r.path, "/")
// started again as a second directory with the same path.
func (b *treeBuilder) add(r scanRec) {
comps := strings.Split(r.path, "/")
dirNames, name := comps[:len(comps)-1], comps[len(comps)-1]
// Keep the open directories that hold this path; complete the rest. node := super
depth := 1 for _, c := range comps[:len(comps)-1] {
for depth < len(b.open) && depth <= len(dirNames) && child := node.dirs[c]
b.names[depth] == dirNames[depth-1] { if child == nil {
depth++ childPath := node.path + "/" + c
// The root directory's path is "/", not empty, and its
// children's paths start with one slash, not two.
switch {
case node == super && c == "":
childPath = "/"
case node == super:
childPath = c
case node.path == "/":
childPath = "/" + c
}
child = &treeNode{path: childPath, parent: node}
if node.dirs == nil {
node.dirs = make(map[string]*treeNode)
}
node.dirs[c] = child
allDirs = append(allDirs, child)
}
node = child
}
if node.files == nil {
node.files = make(map[string]fileSig)
}
sig := fileSig{
size: r.size, head: r.head, tail: r.tail, content: r.content,
}
// A record without a content hash has unknown content (README
// "Database"): give it a signature no other file can share, so
// trees containing it never compare equal. Real hashes are
// hex, so the NUL-prefixed form cannot collide.
if sig.content == "" {
sig.content = "unhashed\x00" + r.path
}
node.files[comps[len(comps)-1]] = sig
} }
b.closeTo(depth) return super, allDirs
for _, c := range dirNames[depth-1:] {
b.openDir(c)
}
dir := b.open[len(b.open)-1]
dir.entries = append(dir.entries, fileEntry(name, r))
dir.fileCount++
dir.totalSize += r.size
}
// openDir starts the directory called name inside the innermost open
// one.
func (b *treeBuilder) openDir(name string) {
parent := b.open[len(b.open)-1]
path := parent.path + "/" + name
// The root directory's path is "/", not empty, and its children's
// paths start with one slash, not two.
switch {
case parent == b.super && name == "":
path = "/"
case parent == b.super:
path = name
case parent.path == "/":
path = "/" + name
}
b.open = append(b.open, &treeNode{path: path, parent: parent})
b.names = append(b.names, name)
}
// closeTo completes the open directories after the first n, innermost
// first: each one's digest is computed and entered in its parent along
// with its totals.
func (b *treeBuilder) closeTo(n int) {
for len(b.open) > n {
last := len(b.open) - 1
dir, name := b.open[last], b.names[last]
b.open, b.names = b.open[:last], b.names[:last]
dir.computeDigest()
dir.parent.entries = append(dir.parent.entries,
"d\x00"+name+"\x00"+string(dir.digest[:]))
dir.parent.fileCount += dir.fileCount
dir.parent.totalSize += dir.totalSize
b.dirs = append(b.dirs, dir)
}
}
// finish completes every open directory and returns the super-root and
// every directory.
func (b *treeBuilder) finish() (*treeNode, []*treeNode) {
b.closeTo(1)
return b.super, b.dirs
}
// fileEntry serializes a file child for its directory's digest: its
// name and its signature (size, head, tail, content); mtime is
// excluded.
func fileEntry(name string, r scanRec) string {
content := r.content
// A record without a content hash has unknown content (README
// "Database"): give it a signature no other file can share, so
// trees containing it never compare equal. Real hashes are hex, so
// the NUL-prefixed form cannot collide.
if content == "" {
content = "unhashed\x00" + r.path
}
return "f\x00" + name + "\x00" + strconv.FormatInt(r.size, 10) +
"\x00" + r.head + "\x00" + r.tail + "\x00" + content
} }
// collectTreeGroups groups directories by digest and returns every // collectTreeGroups groups directories by digest and returns every
@@ -254,22 +190,38 @@ func collectTreeGroups(allDirs []*treeNode, super *treeNode) [][]*treeNode {
return dupes return dupes
} }
// computeDigest sets n's digest and drops its entries. A directory's // compute fills in digest, fileCount, and totalSize for n and all of
// digest is the SHA-256 of its child entries — files serialized with // its descendants. A directory's digest is the SHA-256 of its child
// name and signature, subdirectories with name and recursive digest — // entries — files serialized with name and signature, subdirectories
// sorted byte-lexicographically. Filenames cannot contain NUL or "/", // with name and recursive digest — sorted byte-lexicographically.
// so NUL delimiters are unambiguous. // Filenames cannot contain NUL or "/", so NUL delimiters are
func (n *treeNode) computeDigest() { // unambiguous.
slices.Sort(n.entries) func (n *treeNode) compute() {
entries := make([]string, 0, len(n.dirs)+len(n.files))
for name, sig := range n.files {
entries = append(entries,
"f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+
"\x00"+sig.head+"\x00"+sig.tail+"\x00"+sig.content)
n.fileCount++
n.totalSize += sig.size
}
for name, child := range n.dirs {
child.compute()
entries = append(entries, "d\x00"+name+"\x00"+string(child.digest[:]))
n.fileCount += child.fileCount
n.totalSize += child.totalSize
}
slices.Sort(entries)
h := sha256.New() h := sha256.New()
for _, e := range n.entries { for _, e := range entries {
h.Write([]byte(e)) h.Write([]byte(e))
h.Write([]byte{0}) h.Write([]byte{0})
} }
copy(n.digest[:], h.Sum(nil)) copy(n.digest[:], h.Sum(nil))
n.entries = nil
} }
// suppressed reports whether a duplicate-tree group is non-maximal: its // suppressed reports whether a duplicate-tree group is non-maximal: its
+19 -92
View File
@@ -2,7 +2,6 @@ package main
import ( import (
"bytes" "bytes"
"database/sql"
"slices" "slices"
"testing" "testing"
) )
@@ -31,36 +30,6 @@ func smokeTreeRecs() []scanRec {
} }
} }
// dbTree builds the directory hierarchy from the records in db the way
// trees does, and returns the super-root and every directory.
func dbTree(t *testing.T, db *sql.DB) (*treeNode, []*treeNode) {
t.Helper()
tree := newTreeBuilder()
err := loadFileRows(t.Context(), db, tree.add)
if err != nil {
t.Fatal(err)
}
return tree.finish()
}
// treeOf writes recs into a fresh database and builds the directory
// hierarchy from it the way trees does.
func treeOf(t *testing.T, recs []scanRec) (*treeNode, []*treeNode) {
t.Helper()
db := openTestDB(t)
err := applyChanges(t.Context(), db, recs, nil, nil)
if err != nil {
t.Fatal(err)
}
return dbTree(t, db)
}
// nodeByPath finds the directory node with the given path. // nodeByPath finds the directory node with the given path.
func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode { func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode {
t.Helper() t.Helper()
@@ -91,10 +60,11 @@ func groupPaths(groups [][]*treeNode) [][]string {
return out return out
} }
func TestTreeCounts(t *testing.T) { func TestBuildHierarchyCounts(t *testing.T) {
t.Parallel() t.Parallel()
_, dirs := treeOf(t, smokeTreeRecs()) super, dirs := buildHierarchy(smokeTreeRecs())
super.compute()
d := nodeByPath(t, dirs, "/d") d := nodeByPath(t, dirs, "/d")
if d.fileCount != 6 || d.totalSize != 9300 { if d.fileCount != 6 || d.totalSize != 9300 {
@@ -115,12 +85,12 @@ func TestTreeCounts(t *testing.T) {
} }
} }
func TestTreeRootPath(t *testing.T) { func TestBuildHierarchyRootPath(t *testing.T) {
t.Parallel() t.Parallel()
// The root directory's path is "/", never empty, and its // The root directory's path is "/", never empty, and its
// children's paths start with a single slash. // children's paths start with a single slash.
_, dirs := treeOf(t, []scanRec{{path: "/f"}, {path: "/srv/g"}}) _, dirs := buildHierarchy([]scanRec{{path: "/f"}, {path: "/srv/g"}})
got := make([]string, 0, len(dirs)) got := make([]string, 0, len(dirs))
for _, d := range dirs { for _, d := range dirs {
@@ -135,56 +105,6 @@ func TestTreeRootPath(t *testing.T) {
} }
} }
func TestTreeNamesSortingBeforeSlash(t *testing.T) {
t.Parallel()
// In path order "/a/b-x/f" and "/a/b.txt" come between the file
// "/a/b" and "/a/b/f", because "-" and "." sort before "/". Each
// directory must still be built once, whole, so /a matches /c.
recs := make([]scanRec, 0, 8)
for _, top := range []string{"/a", "/c"} {
for _, p := range []string{"/b", "/b-x/f", "/b.txt", "/b/f"} {
content := "c"
if p == "/b-x/f" {
content = "other"
}
recs = append(recs, scanRec{
size: 1, head: "h", tail: "t", content: content, path: top + p,
})
}
}
super, dirs := treeOf(t, recs)
got := make([]string, 0, len(dirs))
for _, d := range dirs {
got = append(got, d.path)
}
slices.Sort(got)
want := []string{"/", "/a", "/a/b", "/a/b-x", "/c", "/c/b", "/c/b-x"}
if !slices.Equal(got, want) {
t.Fatalf("directory paths = %q, want %q", got, want)
}
groups := collectTreeGroups(dirs, super)
gotGroups := groupPaths(groups)
wantGroups := [][]string{{"/a", "/c"}}
if !slices.EqualFunc(gotGroups, wantGroups, slices.Equal) {
t.Fatalf("groups = %v, want %v", gotGroups, wantGroups)
}
if groups[0][0].fileCount != 4 || groups[0][0].totalSize != 4 {
t.Errorf("group totals: %d files %d bytes, want 4 4",
groups[0][0].fileCount, groups[0][0].totalSize)
}
}
func TestRunTreesEscapesPaths(t *testing.T) { func TestRunTreesEscapesPaths(t *testing.T) {
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs())) t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
@@ -206,7 +126,8 @@ func TestRunTreesEscapesPaths(t *testing.T) {
func TestTreeDigests(t *testing.T) { func TestTreeDigests(t *testing.T) {
t.Parallel() t.Parallel()
_, dirs := treeOf(t, smokeTreeRecs()) super, dirs := buildHierarchy(smokeTreeRecs())
super.compute()
t1 := nodeByPath(t, dirs, "/d/t1") t1 := nodeByPath(t, dirs, "/d/t1")
t2 := nodeByPath(t, dirs, "/d/t2") t2 := nodeByPath(t, dirs, "/d/t2")
@@ -239,7 +160,8 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"}, {size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
} }
_, dirs := treeOf(t, recs) super, dirs := buildHierarchy(recs)
super.compute()
a := nodeByPath(t, dirs, "/r/a") a := nodeByPath(t, dirs, "/r/a")
b := nodeByPath(t, dirs, "/r/b") b := nodeByPath(t, dirs, "/r/b")
@@ -252,7 +174,8 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
func TestCollectTreeGroupsMaximal(t *testing.T) { func TestCollectTreeGroupsMaximal(t *testing.T) {
t.Parallel() t.Parallel()
super, dirs := treeOf(t, smokeTreeRecs()) super, dirs := buildHierarchy(smokeTreeRecs())
super.compute()
groups := collectTreeGroups(dirs, super) groups := collectTreeGroups(dirs, super)
@@ -276,14 +199,16 @@ func TestCollectTreeGroupsDeterministic(t *testing.T) {
recs := smokeTreeRecs() recs := smokeTreeRecs()
super, dirs := treeOf(t, recs) super, dirs := buildHierarchy(recs)
super.compute()
forward := groupPaths(collectTreeGroups(dirs, super)) forward := groupPaths(collectTreeGroups(dirs, super))
reversed := slices.Clone(recs) reversed := slices.Clone(recs)
slices.Reverse(reversed) slices.Reverse(reversed)
superR, dirsR := treeOf(t, reversed) superR, dirsR := buildHierarchy(reversed)
superR.compute()
backward := groupPaths(collectTreeGroups(dirsR, superR)) backward := groupPaths(collectTreeGroups(dirsR, superR))
if !slices.EqualFunc(forward, backward, slices.Equal) { if !slices.EqualFunc(forward, backward, slices.Equal) {
@@ -302,7 +227,8 @@ func TestCollectTreeGroupsSiblings(t *testing.T) {
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"}, {size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
} }
super, dirs := treeOf(t, recs) super, dirs := buildHierarchy(recs)
super.compute()
got := groupPaths(collectTreeGroups(dirs, super)) got := groupPaths(collectTreeGroups(dirs, super))
@@ -324,7 +250,8 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) {
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"}, {size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
} }
super, dirs := treeOf(t, recs) super, dirs := buildHierarchy(recs)
super.compute()
got := groupPaths(collectTreeGroups(dirs, super)) got := groupPaths(collectTreeGroups(dirs, super))
-8
View File
@@ -1,8 +0,0 @@
# THIS IS AN AUTOGENERATED FILE. DO NOT EDIT THIS FILE DIRECTLY.
# yarn lockfile v1
prettier@3.8.1:
version "3.8.1"
resolved "https://registry.yarnpkg.com/prettier/-/prettier-3.8.1.tgz#edf48977cf991558f4fcbd8a3ba6015ba2a3a173"
integrity sha512-UOnG6LftzbdaHZcKoPFtOcCKztrQ57WkHDeRD9t/PTQtmT0NHSeWWepj6pS0z/N7+08BHFDQVUrfmfMRcZwbMg==