Compare commits
70
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
56fd779f7a | ||
|
|
c9c8b1d06c | ||
|
|
7278c354f0 | ||
|
|
8032ea682b | ||
|
|
948fb03630 | ||
|
|
7ff0dbb6e1 | ||
|
|
bccc14ffef | ||
|
|
33f8607e17 | ||
|
|
2eeba3df6f | ||
|
|
cb5dda4f45 | ||
|
|
4847882e46 | ||
|
|
33cf3dd29a | ||
|
|
9abf81535a | ||
|
|
2dd1194f33 | ||
|
|
705c8729ca | ||
|
|
01ff3bb5f0 | ||
|
|
d63d3cc7fc | ||
|
|
c887f80f57 | ||
|
|
c9bf22d483 | ||
|
|
c737490a53 | ||
|
|
09a39ddf37 | ||
|
|
29a65016d0 | ||
|
|
7ac4f6b723 | ||
|
|
337b319542 | ||
|
|
d43c1d31ac | ||
|
|
d4eaf5fed2 | ||
|
|
e6a91711b0 | ||
|
|
5ca68804ac | ||
|
|
3a183aa64b | ||
|
|
47fd4e8def | ||
|
|
99d757c31d | ||
|
|
a5fa600c98 | ||
|
|
9322e8ddee | ||
|
|
a102b8fb06 | ||
|
|
964fc29ed3 | ||
|
|
b8ebe5f578 | ||
|
|
9d06c13777 | ||
|
|
9e924721e6 | ||
|
|
076d82231b | ||
|
|
1a38570301 | ||
|
|
1399249957 | ||
|
|
2a055c0104 | ||
|
|
73841c9989 | ||
|
|
ce6d29dffb | ||
|
|
a7295750ef | ||
|
|
38a01bd27c | ||
|
|
814bdada2b | ||
|
|
3abeacf8ee | ||
|
|
b5f6faa00e | ||
|
|
b14b735c88 | ||
|
|
67bde6226d | ||
|
|
57efa64f18 | ||
|
|
d4b43ebb30 | ||
|
|
b62b4f297f | ||
|
|
340bdbe39e | ||
|
|
a1c3b852c3 | ||
|
|
9f03eb3e2a | ||
|
|
3ecf73c80a | ||
|
|
732fc351d7 | ||
|
|
1e7a519608 | ||
|
|
09ff9b5f30 | ||
|
|
a0f0050ada | ||
|
|
6a15b879de | ||
|
|
dced5cf0d2 | ||
|
|
3ebf98940a | ||
|
|
abea945730 | ||
|
|
d8fbcb32c2 | ||
|
|
e0d578a707 | ||
|
|
90c9ef3546 | ||
|
|
13b9839e73 |
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"worktree": {
|
||||
"bgIsolation": "none"
|
||||
}
|
||||
}
|
||||
+6
-2
@@ -1,8 +1,12 @@
|
||||
.git
|
||||
# .git is sent without its config. Without a VERSION build argument the
|
||||
# stage that compiles runs `git describe --tags --always` on .git, which
|
||||
# does not need .git/config; that file can hold a credential, such as a
|
||||
# password in a remote URL or the token the CI checkout step stores there.
|
||||
.git/config
|
||||
|
||||
.claude
|
||||
.DS_Store
|
||||
sfdupes
|
||||
files.dat
|
||||
*.log
|
||||
*.out
|
||||
*.test
|
||||
|
||||
@@ -6,4 +6,6 @@ jobs:
|
||||
steps:
|
||||
# actions/checkout v4.2.2, 2026-02-22
|
||||
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
|
||||
- run: docker build .
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- run: script/cibuild
|
||||
|
||||
+3
-1
@@ -27,7 +27,9 @@ node_modules/
|
||||
*.log
|
||||
|
||||
# Local scan data
|
||||
files.dat
|
||||
*.sqlite
|
||||
*.sqlite-shm
|
||||
*.sqlite-wal
|
||||
|
||||
# Agent worktrees
|
||||
.claude/worktrees/
|
||||
|
||||
+5
-3
@@ -1,5 +1,9 @@
|
||||
version: "2"
|
||||
|
||||
# Config schema uses the golangci-lint v2 layout (settings live under
|
||||
# linters.settings, not top-level linters-settings) so that the
|
||||
# thresholds below are actually applied by golangci-lint >= v2.
|
||||
|
||||
run:
|
||||
timeout: 5m
|
||||
modules-download-mode: readonly
|
||||
@@ -14,8 +18,7 @@ linters:
|
||||
- wsl # Deprecated, replaced by wsl_v5
|
||||
- wrapcheck # Too verbose for internal packages
|
||||
- varnamelen # Short names like db, id are idiomatic Go
|
||||
|
||||
linters-settings:
|
||||
settings:
|
||||
lll:
|
||||
line-length: 88
|
||||
funlen:
|
||||
@@ -27,6 +30,5 @@ linters-settings:
|
||||
threshold: 100
|
||||
|
||||
issues:
|
||||
exclude-use-default: false
|
||||
max-issues-per-linter: 0
|
||||
max-same-issues: 0
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
node_modules/
|
||||
yarn.lock
|
||||
@@ -0,0 +1,4 @@
|
||||
{
|
||||
"tabWidth": 4,
|
||||
"proseWrap": "always"
|
||||
}
|
||||
+157
-13
@@ -1,33 +1,177 @@
|
||||
# Lint stage — fast feedback on formatting and lint issues
|
||||
# golangci/golangci-lint:v2.12.1, 2026-07-23
|
||||
FROM golangci/golangci-lint@sha256:c9843d374ca80ecbac86081ec4dd7fe2bb6187b03224f59a0cc2f80759e1845b AS lint
|
||||
# golangci/golangci-lint:v2.12.2, 2026-08-07
|
||||
FROM golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240 AS lint
|
||||
WORKDIR /src
|
||||
COPY go.mod go.sum ./
|
||||
RUN go mod download
|
||||
COPY . .
|
||||
RUN make fmt-check
|
||||
RUN make lint
|
||||
|
||||
# Cache-buster for the gate layers, and only for them. Docker
|
||||
# invalidates COPY only when the copied content changes, so on an
|
||||
# unchanged tree the gates below would be served from cache and the
|
||||
# build would exit 0 having run nothing. script/cibuild and
|
||||
# script/docker pass a fresh CHECK_EPOCH on every invocation.
|
||||
#
|
||||
# Two properties this depends on. ARG is per-stage, so the markdown and
|
||||
# build stages below declare it again; one declaration here would leave
|
||||
# their gates cacheable. And each gate RUN must reference the value,
|
||||
# because BuildKit hashes the expanded command: a declared but
|
||||
# unreferenced ARG invalidates nothing.
|
||||
#
|
||||
# It sits below the dependency layers deliberately. Everything above it
|
||||
# (the pinned base image, go mod download) keeps its cache; only the
|
||||
# gates go cold.
|
||||
ARG CHECK_EPOCH
|
||||
|
||||
# The linter is invoked directly here, not through `make lint`. That
|
||||
# target now runs `docker build -f Dockerfile.lint`, and a docker build
|
||||
# cannot run a docker build: routing the gate through make would mean
|
||||
# nesting docker inside this image. Same reason `make check` is gone
|
||||
# from the build stage below, and `make fmt-check` from both stages: it
|
||||
# runs prettier through docker too. Its gofmt half is the step below,
|
||||
# its Markdown half the markdown stage further down. gofmt's output is
|
||||
# assigned to a variable first so that its own exit status, as when it
|
||||
# cannot parse a file, still fails the step.
|
||||
RUN echo "gate gofmt, epoch ${CHECK_EPOCH}" && \
|
||||
files="$(gofmt -s -l .)" && \
|
||||
if [ -n "$files" ]; then \
|
||||
echo "gofmt: files not formatted:" >&2; echo "$files" >&2; exit 1; \
|
||||
fi
|
||||
|
||||
# The FROM above and the one in Dockerfile.lint pin the same linter
|
||||
# twice, and nothing else keeps them in sync; this fails the build when
|
||||
# they disagree. See the script for why it restates neither pin.
|
||||
RUN echo "gate lint-image-pin, epoch ${CHECK_EPOCH}" && \
|
||||
script/verify-lint-image-pin
|
||||
|
||||
# Same config-schema check Dockerfile.lint runs, kept here so this build
|
||||
# gates on exactly what script/lint gates on. It validates against a
|
||||
# schema the pinned binary embeds, so it needs no network.
|
||||
RUN echo "gate config verify, epoch ${CHECK_EPOCH}" && \
|
||||
golangci-lint config verify --config .golangci.yml
|
||||
|
||||
RUN echo "gate lint, epoch ${CHECK_EPOCH}" && \
|
||||
golangci-lint run --config .golangci.yml ./...
|
||||
|
||||
# Prettier stage: the prettier that formats this repository's Markdown,
|
||||
# never installed on a host. script/fmt and script/fmt-check build this
|
||||
# stage alone and run it with the repository mounted on /src. prettier
|
||||
# is installed in /tools so that the repository, mounted or copied onto
|
||||
# /src, cannot hide it.
|
||||
# node:22-alpine, 2026-02-22
|
||||
FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS prettier
|
||||
WORKDIR /tools
|
||||
# yarn.lock pins prettier by hash, and --frozen-lockfile fails rather
|
||||
# than install anything yarn.lock does not name.
|
||||
COPY package.json yarn.lock ./
|
||||
RUN yarn install --frozen-lockfile
|
||||
ENV PATH=/tools/node_modules/.bin:$PATH
|
||||
WORKDIR /src
|
||||
|
||||
# Markdown stage: the Markdown half of `make fmt-check`, as a gate.
|
||||
FROM prettier AS markdown
|
||||
COPY . .
|
||||
# Second per-stage declaration of the gate cache-buster; see the lint
|
||||
# stage above.
|
||||
ARG CHECK_EPOCH
|
||||
RUN echo "gate prettier, epoch ${CHECK_EPOCH}" && \
|
||||
prettier --check '**/*.md' --tab-width 4 --prose-wrap always
|
||||
|
||||
# Build stage
|
||||
# golang:1.25-alpine, 2026-07-23
|
||||
FROM golang@sha256:56961d79ea8129efddcc0b8643fd8a5416b4e6228cfd477e3fd61deb2672c587 AS builder
|
||||
|
||||
RUN apk add --no-cache make
|
||||
# We never build or run as root. Create an unprivileged user and point
|
||||
# HOME and the build cache at its home so go build and go test can write
|
||||
# it when we drop to it below. $GOPATH/bin is deliberately not on PATH:
|
||||
# script/bootstrap no longer `go install`s anything (the linter runs
|
||||
# from a pinned image, never from a host install), so nothing lands
|
||||
# there and adding it would only widen what this image resolves.
|
||||
#
|
||||
# The module cache is kept outside that home, at the base image's
|
||||
# default /go/pkg/mod, and belongs to root: script/bootstrap fills it as
|
||||
# root. Do not move it into the home and hand it over with `chown -R`:
|
||||
# that walks every file in it, which took from about 80 s to over ten
|
||||
# minutes on a shared host, depending on load.
|
||||
RUN adduser -D -u 1000 builder
|
||||
ENV HOME=/home/builder
|
||||
ENV GOPATH=/home/builder/go
|
||||
ENV GOMODCACHE=/go/pkg/mod
|
||||
ENV GOCACHE=/home/builder/.cache/go-build
|
||||
|
||||
WORKDIR /src
|
||||
|
||||
# Reuse the linter binary from the lint stage; the copy also forces
|
||||
# BuildKit to complete linting before this stage proceeds.
|
||||
COPY --from=lint /usr/bin/golangci-lint /usr/local/bin/golangci-lint
|
||||
# No-op file copies whose only purpose is the build-graph edge: they are
|
||||
# what make this stage depend on the lint and markdown stages, and so
|
||||
# what forces BuildKit to finish gofmt, the pin guard, lint and prettier
|
||||
# before compilation and tests start. Remove one and the fail-fast
|
||||
# design dies silently — the build stops gating on that stage and still
|
||||
# exits 0. The first replaces a copy of the linter binary itself, which
|
||||
# is no longer wanted here: nothing in this stage runs the linter,
|
||||
# because `make lint` is now a docker build and a docker build cannot
|
||||
# run inside one.
|
||||
COPY --from=lint /src/go.sum /dev/null
|
||||
COPY --from=markdown /src/go.sum /dev/null
|
||||
|
||||
# Install development prerequisites the same way a developer does,
|
||||
# rather than duplicating the installs inline. Only script/ and the
|
||||
# dependency manifests are copied first, nothing else, so this layer
|
||||
# stays cached until the scripts or the dependencies change — bootstrap
|
||||
# ends in `go mod download`, which is why there is no separate
|
||||
# invocation of it here.
|
||||
COPY script/ script/
|
||||
COPY go.mod go.sum ./
|
||||
RUN go mod download
|
||||
COPY . .
|
||||
RUN script/bootstrap
|
||||
|
||||
# Fail the build unless the branch is green.
|
||||
RUN make check
|
||||
# Hand builder only what it writes to, without walking the module cache.
|
||||
# This layer stays cached with bootstrap.
|
||||
# - /src itself: make build writes the binary into it, and git refuses
|
||||
# a repository whose top directory belongs to another user.
|
||||
# - the module cache's cache/download directory itself, not what is in
|
||||
# it: Go only reads the downloaded modules, but make build saves its
|
||||
# lookup of this module's own version from git there, in a new
|
||||
# directory named after the module path.
|
||||
# - builder's home: the go commands bootstrap ran as root left Go's
|
||||
# telemetry files there, a few small files.
|
||||
RUN chown builder:builder /src /go/pkg/mod/cache/download && \
|
||||
chown -R builder:builder /home/builder
|
||||
|
||||
RUN make build
|
||||
# The sources are handed to builder as they are copied, so no layer has
|
||||
# to walk them. Then drop root before running any checks or builds.
|
||||
COPY --chown=builder:builder . .
|
||||
USER builder
|
||||
|
||||
# Fail the build unless the branch is green. Runs as non-root so the
|
||||
# permission-denied test paths are exercised legitimately (root would
|
||||
# bypass the chmod(0) the tests rely on).
|
||||
#
|
||||
# The gate is `make test`, not `make check`: that aggregate runs
|
||||
# `script/lint` and `script/fmt-check`, which both run docker, and
|
||||
# nothing inside an image build may shell out to docker. Lint and the
|
||||
# format checks are not skipped by this — they ran in the lint and
|
||||
# markdown stages above, which this stage's COPY --from lines make
|
||||
# prerequisites. `make`, not the script directly, because the Makefile's
|
||||
# `export CGO_ENABLED = 0` applies only to what it invokes.
|
||||
#
|
||||
# Third per-stage declaration of the gate cache-buster; see the lint
|
||||
# stage above for why one is not enough. It is placed after USER so the
|
||||
# drop to the unprivileged user still happens before the checks run.
|
||||
ARG CHECK_EPOCH
|
||||
RUN echo "gate test, epoch ${CHECK_EPOCH}" && make test
|
||||
|
||||
# The version stamped into the binary: the VERSION build argument when
|
||||
# one is given, otherwise `git describe --tags --always` of the .git in
|
||||
# the build context (git is installed by script/bootstrap above). A
|
||||
# context that carries .git and still yields no version fails the build;
|
||||
# with neither, as from a source tarball, it is "dev".
|
||||
ARG VERSION
|
||||
RUN version="${VERSION:-$(git describe --tags --always || echo dev)}"; \
|
||||
if [ -e .git ] && { [ -z "$version" ] || [ "$version" = dev ] || \
|
||||
[ "$version" = unknown ]; }; then \
|
||||
echo "no version could be derived although the build context carries .git" >&2; \
|
||||
exit 1; \
|
||||
fi; \
|
||||
make build VERSION="$version"
|
||||
|
||||
# Runtime stage
|
||||
# alpine:3.22, 2026-07-23
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
# Lint-only image: this is how the linter runs, everywhere. The repo is
|
||||
# COPYed into the pinned golangci-lint image and the linter runs as a
|
||||
# build step, so a successful build IS a clean lint. golangci-lint is
|
||||
# never installed on a host — one toolchain, pinned by digest, identical
|
||||
# on a laptop and in CI — and this works even when the docker daemon is
|
||||
# remote and bind mounts are impossible.
|
||||
#
|
||||
# script/lint builds this file. It is a separate image from the lint
|
||||
# stage of the main Dockerfile because script/lint must not depend on
|
||||
# the rest of that build; the two FROM lines are kept identical by
|
||||
# script/verify-lint-image-pin, run as a gate below.
|
||||
# golangci/golangci-lint:v2.12.2, 2026-08-07
|
||||
FROM golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240
|
||||
|
||||
WORKDIR /src
|
||||
|
||||
# Dependency layers first, so they stay cached across lint runs.
|
||||
COPY go.mod go.sum ./
|
||||
RUN go mod download
|
||||
|
||||
COPY . .
|
||||
|
||||
# Cache-buster for the gate layers, and only for them. Caching of the
|
||||
# lint run is waived by ruling: COPY is invalidated only by changed
|
||||
# content, so on an unchanged tree the gates below would be served from
|
||||
# cache and this build would exit 0 in under a second having run no
|
||||
# linter at all. That exact false green has bitten this repo twice
|
||||
# already (#32, #39). script/lint passes a fresh value on every
|
||||
# invocation.
|
||||
#
|
||||
# Each gate RUN must reference the value, because BuildKit hashes the
|
||||
# expanded command and not the ARG declaration: a declared but
|
||||
# unreferenced ARG invalidates nothing. The ARG sits below the
|
||||
# dependency layers deliberately — everything above it keeps its cache,
|
||||
# only the gates go cold.
|
||||
ARG CHECK_EPOCH
|
||||
|
||||
# The linter version is pinned in two places, here and in the main
|
||||
# Dockerfile's lint stage. Nothing else keeps them in sync, so a
|
||||
# half-applied bump is a build failure; see the script.
|
||||
RUN echo "gate lint-image-pin, epoch ${CHECK_EPOCH}" && \
|
||||
script/verify-lint-image-pin
|
||||
|
||||
# Validates .golangci.yml against golangci-lint's JSON schema. The
|
||||
# concern about this step was that it fetches that schema over a live,
|
||||
# unpinned HTTPS call; measured on the pinned image, it does not. The
|
||||
# binary carries the schema for its own version, so under
|
||||
# `--network none` this both passes on a valid config and still rejects
|
||||
# an invalid one with the jsonschema error. That holds for the gate
|
||||
# steps generally — none of them makes a network call — but not for
|
||||
# this build as a whole: `go mod download` above needs the network on a
|
||||
# cold cache, and under `--network none` a first build fails there
|
||||
# before reaching any gate. That layer stays cached, so only a warm
|
||||
# cache lints offline, until go.mod or go.sum changes.
|
||||
RUN echo "gate config verify, epoch ${CHECK_EPOCH}" && \
|
||||
golangci-lint config verify --config .golangci.yml
|
||||
|
||||
RUN echo "gate lint, epoch ${CHECK_EPOCH}" && \
|
||||
golangci-lint run --config .golangci.yml ./...
|
||||
@@ -6,43 +6,44 @@ BINARY := sfdupes
|
||||
VERSION := $(shell git describe --tags --always --dirty 2>/dev/null || echo dev)
|
||||
LDFLAGS := -X main.Version=$(VERSION)
|
||||
|
||||
.PHONY: all build test lint fmt fmt-check check docker hooks clean
|
||||
.PHONY: sfdupes build bootstrap setup test lint fmt fmt-check check docker hooks clean
|
||||
|
||||
all: check build
|
||||
# Standard targets are thin shims; the implementations live in script/
|
||||
# per the scripts-to-rule-them-all pattern.
|
||||
|
||||
build:
|
||||
# Default target: build the binary. Phony so go build (which has its
|
||||
# own build cache) always decides what to recompile.
|
||||
sfdupes:
|
||||
go build -ldflags "$(LDFLAGS)" -o $(BINARY)
|
||||
|
||||
build: sfdupes
|
||||
|
||||
bootstrap:
|
||||
@script/bootstrap
|
||||
|
||||
setup:
|
||||
@script/setup
|
||||
|
||||
test:
|
||||
@go test -timeout 30s -cover ./... || \
|
||||
{ echo "--- Rerunning with -v for details ---"; \
|
||||
go test -timeout 30s -v ./...; exit 1; }
|
||||
@script/test
|
||||
|
||||
lint:
|
||||
golangci-lint run --config .golangci.yml ./...
|
||||
@script/lint
|
||||
|
||||
fmt:
|
||||
gofmt -s -w .
|
||||
@script/fmt
|
||||
|
||||
fmt-check:
|
||||
@files="$$(gofmt -l -s .)"; if [ -n "$$files" ]; then \
|
||||
echo "gofmt: files not formatted:"; echo "$$files"; exit 1; fi
|
||||
@script/fmt-check
|
||||
|
||||
check: test lint fmt-check
|
||||
check:
|
||||
@script/check
|
||||
|
||||
docker:
|
||||
docker build -t $(BINARY) .
|
||||
|
||||
# Hooks are shared between the main checkout and all worktrees, so
|
||||
# resolve the common git dir instead of assuming .git is a directory.
|
||||
HOOKS_DIR := $(shell git rev-parse --git-common-dir)/hooks
|
||||
@script/docker
|
||||
|
||||
hooks:
|
||||
@printf '#!/bin/sh\nset -e\n' > $(HOOKS_DIR)/pre-commit
|
||||
@printf 'go mod tidy\ngo fmt ./...\n' >> $(HOOKS_DIR)/pre-commit
|
||||
@printf 'git diff --exit-code -- go.mod go.sum || { echo "go mod tidy changed files; please stage and retry"; exit 1; }\n' >> $(HOOKS_DIR)/pre-commit
|
||||
@printf 'make check\n' >> $(HOOKS_DIR)/pre-commit
|
||||
@chmod +x $(HOOKS_DIR)/pre-commit
|
||||
@script/install-precommit
|
||||
|
||||
clean:
|
||||
rm -f $(BINARY) files.dat
|
||||
rm -f $(BINARY)
|
||||
|
||||
+60
-20
@@ -34,10 +34,46 @@ style conventions are in separate documents:
|
||||
every file before committing. There are zero exceptions to this rule.
|
||||
|
||||
- Every repo with software must have a root `Makefile` with these targets:
|
||||
`make test`, `make lint`, `make fmt` (writes), `make fmt-check` (read-only),
|
||||
`make check` (prereqs: `test`, `lint`, `fmt-check`), `make docker`, and
|
||||
`make hooks` (installs pre-commit hook). A model Makefile is at
|
||||
`https://git.eeqj.de/sneak/prompts/raw/branch/main/Makefile`.
|
||||
`make bootstrap`, `make setup`, `make test`, `make lint`, `make fmt` (writes),
|
||||
`make fmt-check` (read-only), `make check` (runs `test`, `lint`, `fmt-check`),
|
||||
`make docker`, and `make hooks` (installs pre-commit hook). A model Makefile
|
||||
is at `https://git.eeqj.de/sneak/prompts/raw/branch/main/Makefile`.
|
||||
|
||||
- Repos follow the
|
||||
[Scripts to Rule Them All](https://github.com/github/scripts-to-rule-them-all)
|
||||
pattern: the implementation of each Makefile target lives in an executable
|
||||
script in `script/` (`script/bootstrap`, `script/setup`, `script/test`,
|
||||
`script/lint`, `script/fmt`, `script/fmt-check`, `script/check`,
|
||||
`script/docker`), and the Makefile targets are thin shims that call them. The
|
||||
scripts must be POSIX sh (`#!/bin/sh`, `set -eu`, no bashisms) so they run in
|
||||
minimal containers (e.g. alpine images have no bash); locate the repo root
|
||||
with `$(cd "$(dirname "$0")/.." && pwd -P)` and `cd` there before acting. From
|
||||
the standard's canonical set we use `bootstrap`, `setup` (make the repo ready
|
||||
for development after a fresh clone: runs `bootstrap`, then
|
||||
`install-precommit`, plus any repo-specific initialization), `test`, and
|
||||
`cibuild`. `script/bootstrap` installs all dependencies idempotently and
|
||||
assumes nothing is present: base tools come from nix, apt, brew, or apk
|
||||
(detected in that order; apt runs noninteractive). For node it uses the
|
||||
installed node if present; otherwise it installs a PINNED node version via
|
||||
nvm, first installing nvm itself if missing — from a hash-verified GitHub
|
||||
release archive (never `curl | sh`), with bash installed as an explicit
|
||||
prerequisite since nvm requires bash. yarn is then pinned via
|
||||
`corepack prepare yarn@<version> --activate`. Never install "latest" or "lts";
|
||||
always exact versions. `script/cibuild` runs the CI build: it changes to the
|
||||
repo root and runs `docker build .`; the Gitea workflow calls it. Four further
|
||||
scripts are our own extensions to the standard: `script/check` runs
|
||||
`script/test`, `script/lint`, and `script/fmt-check`; `script/precommit` is
|
||||
what the git pre-commit hook runs, and it calls `script/check`;
|
||||
`script/install-precommit` installs the git pre-commit hook (the `make hooks`
|
||||
target shims to it); and `script/projectname` (literally that filename) simply
|
||||
outputs the project's name. Scripts that need the name call
|
||||
`script/projectname` — e.g. `script/docker` assembles its image tag from it —
|
||||
so those scripts stay byte-identical across all repos. Repo-type-specific
|
||||
pre-commit extras (e.g. `go mod tidy` verification in Go repos) belong in
|
||||
`script/precommit`, not in the hook itself. Model scripts are at
|
||||
`https://git.eeqj.de/sneak/prompts/raw/branch/main/script/<name>`. The README
|
||||
must document the provided scripts in an **Entrypoints** section (see the
|
||||
README requirements below).
|
||||
|
||||
- Always use Makefile targets (`make fmt`, `make test`, `make lint`, etc.)
|
||||
instead of invoking the underlying tools directly. The Makefile is the single
|
||||
@@ -57,7 +93,11 @@ style conventions are in separate documents:
|
||||
as a build step so the build fails if the branch is not green. For non-server
|
||||
repos, the Dockerfile should bring up a development environment and run
|
||||
`make check`. For server repos, `make check` should run as an early build
|
||||
stage before the final image is assembled.
|
||||
stage before the final image is assembled. Dockerfiles install development
|
||||
prerequisites by running `script/bootstrap` rather than duplicating installs
|
||||
inline; COPY `script/` and the dependency manifests (`package.json` +
|
||||
`yarn.lock`, `go.mod` + `go.sum`, etc.) before running it so the bootstrap
|
||||
layer stays cached until dependencies change.
|
||||
|
||||
- **Dockerfiles must use a separate lint stage for fail-fast feedback.** Go
|
||||
repos use a multistage build where linting runs in an independent stage based
|
||||
@@ -127,8 +167,9 @@ style conventions are in separate documents:
|
||||
artifacts or heavier dependencies.
|
||||
|
||||
- Every repo should have a Gitea Actions workflow (`.gitea/workflows/`) that
|
||||
runs `docker build .` on push. Since the Dockerfile already runs `make check`,
|
||||
a successful build implies all checks pass.
|
||||
runs `script/cibuild` (which runs `docker build .`) on push. Since the
|
||||
Dockerfile already runs `make check`, a successful build implies all checks
|
||||
pass.
|
||||
|
||||
- Use platform-standard formatters: `black` for Python, `prettier` for
|
||||
JS/CSS/Markdown/HTML, `go fmt` for Go. Always use default configuration with
|
||||
@@ -136,9 +177,11 @@ style conventions are in separate documents:
|
||||
Markdown (hard-wrap at 80 columns). Documentation and writing repos (Markdown,
|
||||
HTML, CSS) should also have `.prettierrc` and `.prettierignore`.
|
||||
|
||||
- Pre-commit hook: `make check` if local testing is possible, otherwise
|
||||
`make lint && make fmt-check`. The Makefile should provide a `make hooks`
|
||||
target to install the pre-commit hook.
|
||||
- Pre-commit hook: runs `script/precommit`, which calls `script/check`. If local
|
||||
testing is not possible in the repo, `script/precommit` may skip `script/test`
|
||||
and run only `script/lint` and `script/fmt-check`. The hook is installed by
|
||||
`script/install-precommit`; the Makefile must provide a `make hooks` target
|
||||
that shims to it.
|
||||
|
||||
- All repos with software must have tests that run via the platform-standard
|
||||
test framework (`go test`, `pytest`, `jest`/`vitest`, etc.). If no meaningful
|
||||
@@ -297,6 +340,10 @@ style conventions are in separate documents:
|
||||
"µPaaS is an MIT-licensed Go web application by @sneak that receives
|
||||
git-frontend webhooks and deploys applications via Docker in realtime."
|
||||
- **Getting Started**: Copy-pasteable install/usage code block.
|
||||
- **Entrypoints**: Opens by stating that the repo adheres to the
|
||||
[Scripts to Rule Them All](https://github.com/github/scripts-to-rule-them-all)
|
||||
standard (with that link), then documents each provided `script/`
|
||||
entrypoint and its purpose.
|
||||
- **Rationale**: Why does this exist?
|
||||
- **Design**: How is the program structured?
|
||||
- **TODO**: Update meticulously, even between commits. When planning, put
|
||||
@@ -326,16 +373,6 @@ style conventions are in separate documents:
|
||||
- All repos should have an `.editorconfig` enforcing the project's indentation
|
||||
settings.
|
||||
|
||||
- **Claude Code repo memory is versioned in the repo**, not left only in
|
||||
`~/.claude` on one machine. Each memory is one file at
|
||||
`.claude/memory/<memory>.md`, and every memory file must be `@`-imported
|
||||
from `.claude/CLAUDE.md` (one `- @memory/<memory>.md` list line per file;
|
||||
relative import paths resolve against `.claude/`, and Claude Code expands
|
||||
the imports into context at session launch). When adding a memory, add both
|
||||
the file and its import line. A root `MEMORY.md` is a violation — Claude
|
||||
Code never auto-loads it; split it into `.claude/memory/` files. Repos with
|
||||
no memories yet need no `.claude/` scaffolding.
|
||||
|
||||
- Avoid putting files in the repo root unless necessary. Root should contain
|
||||
only project-level config files (`README.md`, `Makefile`, `Dockerfile`,
|
||||
`LICENSE`, `.gitignore`, `.editorconfig`, `REPO_POLICIES.md`, and
|
||||
@@ -361,6 +398,9 @@ style conventions are in separate documents:
|
||||
- `README.md`, `.git`, `.gitignore`, `.editorconfig`
|
||||
- `LICENSE`, `REPO_POLICIES.md` (copy from the `prompts` repo)
|
||||
- `Makefile`
|
||||
- `script/` entrypoints (`bootstrap`, `setup`, `projectname`, `test`,
|
||||
`lint`, `fmt`, `fmt-check`, `check`, `docker`, `cibuild`, `precommit`,
|
||||
`install-precommit`)
|
||||
- `Dockerfile`, `.dockerignore`
|
||||
- `.gitea/workflows/check.yml`
|
||||
- Go: `go.mod`, `go.sum`, `.golangci.yml`
|
||||
|
||||
@@ -1,81 +1,479 @@
|
||||
# Workflow
|
||||
|
||||
- take an issue from the `1.0.0` milestone on the tracker; work not yet on the
|
||||
tracker gets filed as an issue first
|
||||
- branch (from `main`)
|
||||
- do the work in Next Step
|
||||
- move Next Step to the top of Completed Steps
|
||||
- move the top item of Future Steps into Next Step
|
||||
- commit (`TODO.md` changes in the same commit as the work)
|
||||
- merge to `main` if the branch is not protected, otherwise open a PR
|
||||
- push
|
||||
- do the work, with tests, in small focused commits
|
||||
- record it at the top of Completed Steps (`TODO.md` changes in the same commit
|
||||
as the work)
|
||||
- push the branch and open a PR whose title ends with ` (closes #N)`
|
||||
- an independent review gates the merge; every finding is addressed or
|
||||
explicitly rebutted on the PR
|
||||
- merge to `main` once the review passes
|
||||
|
||||
# Status
|
||||
|
||||
- pre-1.0
|
||||
- the Gitea tracker is authoritative for the pre-1.0 backlog: the open issues
|
||||
under the `1.0.0` milestone are what remains before the tag, and this file
|
||||
records history and process, not the queue
|
||||
|
||||
# Next Step
|
||||
|
||||
- add a remote on git.eeqj.de and push (`main` plus tags)
|
||||
- take the next issue from the `1.0.0` milestone on the tracker:
|
||||
https://git.eeqj.de/sneak/sfdupes/milestone/17 — the milestone is the source
|
||||
of truth for what is left before 1.0.0. Individual issues are deliberately not
|
||||
restated here; a copy in this file drifts out of date the moment the tracker
|
||||
moves
|
||||
|
||||
# Completed Steps
|
||||
|
||||
- `make fmt` and `make fmt-check` run prettier over all Markdown, in Docker, and
|
||||
CI checks it; all Markdown reformatted (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/19)
|
||||
|
||||
- `scan` rejects `--workers` below 1 as a usage error instead of running
|
||||
single-threaded (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/10)
|
||||
|
||||
- a test fails when either walk cancellation check in `scan.go` is removed
|
||||
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/81)
|
||||
|
||||
- test that `scan` refuses a database with another schema version (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/64)
|
||||
|
||||
- correct four inaccurate comments in `cancel_test.go` and rename
|
||||
`walkCancelInFlightDirs` to `walkCancelInFlightFiles` (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/33)
|
||||
|
||||
- test the `-x` filesystem-boundary rules in `subdirJob` (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/17)
|
||||
|
||||
- `scan` creates the schema in one transaction; a version-0 database with a
|
||||
`files` table is refused with a clear schema-version error (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/11)
|
||||
|
||||
- README documents install, Docker, a daily cron scan and how to read and check
|
||||
the reports (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/54)
|
||||
|
||||
- the `Dockerfile` build stage keeps the Go module cache out of `builder`'s home
|
||||
and copies the sources with `--chown`, so no `chown -R` walks them
|
||||
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/43)
|
||||
|
||||
- `--version` prints `sfdupes VERSION` to stdout; README documents it and
|
||||
`--help` (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/15)
|
||||
|
||||
- `scan` stops cleanly on `SIGINT` or `SIGTERM`: commits what it has hashed,
|
||||
deletes nothing more, exits 1 (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/5)
|
||||
|
||||
- `report` and `trees` stream the records instead of holding them all in memory;
|
||||
the schema gains the `files_signature` index (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/14)
|
||||
|
||||
- progress prints at once on a non-terminal, uses a real terminal test, and
|
||||
prints warnings through a spinner instead of racing its redraw (2026-10-03,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/13)
|
||||
|
||||
- warn about and skip symlink, socket, FIFO, device and `.zfs` operands, keeping
|
||||
the records beneath them (2026-10-03,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/9)
|
||||
|
||||
- `scan` holds a lock on a lock file beside the database for its whole run, so a
|
||||
second `scan` fails at once with exit 1 (2026-10-03,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/53)
|
||||
|
||||
- test stdout write failures in `report` and `trees`; README states that
|
||||
`| head` ends sfdupes by `SIGPIPE` and `>&-` writes to `/dev/null`
|
||||
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/30)
|
||||
|
||||
- `report` and `trees` open the database read-only, and `scan` leaves it out of
|
||||
WAL mode, so reading needs only read access (2026-10-03, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/8)
|
||||
|
||||
- escape tabs, newlines, carriage returns and backslashes in report, trees and
|
||||
warning paths; the root directory's path is `/` (2026-10-03,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/7)
|
||||
|
||||
- stamp the git tag or short commit in a plain `docker build .` instead of `dev`
|
||||
(2026-10-02, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now sends
|
||||
`.git`, without `.git/config`, and the `Dockerfile` build stage takes the
|
||||
`VERSION` build argument when one is given, otherwise
|
||||
`git describe --tags --always` of that `.git`. The build fails if the context
|
||||
carries `.git` and the version still comes out empty, `dev` or `unknown`. The
|
||||
CI checkout step fetches the full history (`fetch-depth: 0`) so CI sees the
|
||||
tag and stamps the same value as `make build`.
|
||||
|
||||
- replace the 1 KiB end-window sampling with the head/tail plus content-hash
|
||||
ladder (2026-09-22, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is hashed in
|
||||
full and compared directly, with no end-window step — its `head`, `tail`, and
|
||||
`content` all hold the whole-file hash. A file at 10 MiB or above gets only
|
||||
the 64 KiB `head` and `tail` in the hash phase; a new content phase, after the
|
||||
update phase, reads it for its `content` hash — the whole file below 50 MiB,
|
||||
gigabyte-spaced 1 MiB samples at or above — only when its size, `head`, and
|
||||
`tail` match another record's, from the same scan or stored by an earlier one,
|
||||
so a stored file gains its content hash when it gains a match. A file that is
|
||||
gone or has changed since its record was written is not read. The `content`
|
||||
column is part of the version 1 schema. `report` and `trees` group by the
|
||||
extended signature and leave out any record without a `content` hash, so the
|
||||
ladder is applied across the whole database. README "Duplicate detection"
|
||||
documents every rung including the probabilistic large-file path.
|
||||
|
||||
- remove the dead `files.dat` references from `Makefile`, `.gitignore` and
|
||||
`.dockerignore` (2026-09-21, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/22)
|
||||
|
||||
- fix the lint-image pin comments and `FROM` form in `Dockerfile` and
|
||||
`Dockerfile.lint` (2026-08-10, branch `next`, closes
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/25): dropped the false
|
||||
`(Debian-based)` parenthetical (v2.12.1 was Debian too) and the redundant tag,
|
||||
so both pins are the policy `# image:vX.Y.Z, YYYY-MM-DD` comment over a bare
|
||||
`FROM image@sha256:...`. Digest unchanged. `script/verify-lint-image-pin`
|
||||
parses those `FROM` lines and still matches the tagless form; its advice line
|
||||
lost the now meaningless "tag and digest". With no tag in either reference, a
|
||||
tag-only disagreement no longer exists — a one-sided tag is caught as a plain
|
||||
mismatch.
|
||||
|
||||
- run all linting in Docker via `Dockerfile.lint` and `script/lint` (2026-08-10,
|
||||
branch `next`, closes https://git.eeqj.de/sneak/sfdupes/issues/46): per the
|
||||
owner ruling, the linter runs inside a container invoked through the `script/`
|
||||
entrypoint and is never installed on a host. New root `Dockerfile.lint` COPYs
|
||||
the repo into the digest-pinned `golangci/golangci-lint:v2.12.2` image and
|
||||
runs `golangci-lint config verify` and `golangci-lint run` as build steps, so
|
||||
a successful build IS a clean lint; `script/lint` is reduced to building it.
|
||||
`script/bootstrap` loses the `go install`, the pin constants, the version
|
||||
parser and `verify_golangci_lint` outright rather than hardening them — with
|
||||
nothing linting on the host, the `$GOPATH/bin` versus `PATH` problem that
|
||||
motivated them has no subject — and now warns rather than fails when `docker`
|
||||
is absent. Two traps handled. A lint build on an unchanged tree returns
|
||||
success in well under a second having run no linter, which is
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/32 and
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/39 again, so `Dockerfile.lint`
|
||||
carries `ARG CHECK_EPOCH` referenced inside every gate `RUN` (BuildKit hashes
|
||||
the expanded command, not the declaration) and `script/lint` passes
|
||||
`"$(date +%s)-$$"` — the PID matters because two lint runs land inside the
|
||||
same second easily. And nothing inside an image build may shell out to docker,
|
||||
so the main `Dockerfile`'s lint stage now invokes `golangci-lint` directly
|
||||
instead of `make lint`, and its build stage runs `make test` and
|
||||
`make fmt-check` instead of the `make check` aggregate (`make`, not the
|
||||
scripts bare, because the Makefile's `export CGO_ENABLED = 0` only reaches
|
||||
what it invokes). `COPY --from=lint` `/usr/bin/golangci-lint` is replaced by
|
||||
`COPY --from=lint /src/go.sum /dev/null`: the copied binary was the only edge
|
||||
forcing BuildKit to finish linting before the build stage starts, and dropping
|
||||
it without replacing the edge would have ended fail-fast linting silently
|
||||
under a still-green build. That is canonical `REPO_POLICIES.md:107`'s ordering
|
||||
edge, restored. `ENV PATH=/home/builder/go/bin:$PATH` is gone with the
|
||||
`go install` that justified it. `script/verify-linter-pin` is retired, deleted
|
||||
along with its README entry, because both of its subjects ceased to exist in
|
||||
the same change: it compared a linter binary against `GOLANGCI_LINT_VERSION`
|
||||
in `script/bootstrap`, and there is now neither a binary crossing between
|
||||
stages nor a version pin in bootstrap. The drift it guarded has not gone away,
|
||||
it has moved — the linter is still pinned twice, now as the `FROM` line of
|
||||
`Dockerfile.lint` and the `FROM` line of the `Dockerfile` lint stage, with
|
||||
nothing syncing them, which is exactly what
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/42 made a build failure. Its
|
||||
replacement is one new `script/verify-lint-image-pin`, run as a gate in both
|
||||
files, which compares the two references to each other and deliberately
|
||||
restates neither: a hardcoded expected digest would be a third copy and the
|
||||
same drift one file further out. `golangci-lint config verify` is included per
|
||||
the ruling, and the concern about its unpinned live HTTPS schema fetch was
|
||||
measured rather than assumed — under `--network none` the pinned binary both
|
||||
passes a valid config and rejects an invalid one with the jsonschema error, so
|
||||
it validates from an embedded schema and makes no network call of its own. The
|
||||
README scopes that to the gate steps rather than to linting as a whole:
|
||||
`Dockerfile.lint` runs `go mod download` above them, so a cold cache still
|
||||
needs the network and only a warm one lints offline. Verified: `make lint`
|
||||
green with every `PATH` directory containing a `golangci-lint` removed
|
||||
(`/home/user/go/bin`, `/home/user/.local/bin`, `/usr/local/bin`;
|
||||
`command -v golangci-lint` empty); two consecutive `script/lint` runs on an
|
||||
untouched tree both executed the linter, 27.7s and 28.7s in the lint step
|
||||
under distinct epochs with the `COPY . .` layer `CACHED` above them, at 42.2s
|
||||
and 41.8s wall clock — the no-cache rule was not weakened to shorten that.
|
||||
Negative control: a planted `var unusedIssue46Sentinel = 1` failed
|
||||
`script/lint` with
|
||||
`report.go:173:5: var unusedIssue46Sentinel is unused (unused)`, and failed
|
||||
`make docker` at `[lint 9/9]` with the build stage stopped at `[builder 3/12]`
|
||||
— `COPY --from=lint`, `script/bootstrap`, the test gate and `make build` all
|
||||
zero occurrences — then reverted clean. The drift guard fails on a tag-only
|
||||
disagreement, on a digest-only disagreement, and on an unreadable reference,
|
||||
naming both sides. `make docker` green in 5m35s with all six gates executing
|
||||
under one epoch (lint 37.6s, test 25.2s reporting
|
||||
`ok sneak.berlin/go/sfdupes 1.938s coverage: 88.5%`, not `(cached)`). The
|
||||
non-root quirk still holds: in the builder image with the Go test cache off,
|
||||
`--user 0:0` fails `TestScanHardlinkRunFailsTogether` (exit 1) where the
|
||||
unprivileged user passes (exit 0). Noted for follow-up, not fixed here:
|
||||
`golangci-lint` warns that the `gomodguard` linter is deprecated since v2.12.0
|
||||
in favour of `gomodguard_v2`.
|
||||
|
||||
- install the Docker build stage's prerequisites by running `script/bootstrap`
|
||||
instead of `apk add --no-cache make` inline (2026-08-09, branch
|
||||
`dockerfile-bootstrap`, closes #42): canonical `REPO_POLICIES.md:97` requires
|
||||
it, and the inline install left the build stage maintaining its own notion of
|
||||
the toolchain — exactly the divergence #24 exists to close, one layer down.
|
||||
The stage now copies `script/` plus `go.mod`/`go.sum` and runs
|
||||
`script/bootstrap`, which ends in `go mod download`, so the separate
|
||||
invocation of that is gone. `COPY --from=lint /usr/bin/golangci-lint` stays,
|
||||
and moves above the bootstrap layer. It is the only edge making this stage
|
||||
depend on the lint stage, so deleting it as redundant would end fail-fast
|
||||
linting silently. Letting bootstrap install its own linter here would have
|
||||
reintroduced the second toolchain and paid for a from-source build of it. What
|
||||
makes the two stages provably one toolchain rather than two that happen to
|
||||
agree is a new `script/verify-linter-pin`, run in the build stage on the
|
||||
binary that arrives from the lint stage, before bootstrap: it fails the build
|
||||
naming both versions unless that binary is the version `script/bootstrap`
|
||||
pins. Bootstrap's own check could not serve that purpose — it reinstalls its
|
||||
pin from source and then verifies whatever `PATH` resolves, so drift
|
||||
self-heals silently and a lint stage image bumped on its own would lint at the
|
||||
new version while `make check` ran at the old one, green. The linter version
|
||||
is pinned in two independent places (the lint stage image digest and
|
||||
`GOLANGCI_LINT_VERSION`) and nothing else keeps them in sync, so a
|
||||
half-applied bump is now a build failure. The pin is read out of
|
||||
`script/bootstrap`, which stays the single source of truth; a pin that cannot
|
||||
be read is a hard failure, not a skip. The check needs no `CHECK_EPOCH`: its
|
||||
only inputs are the copied binary and `script/`, so Docker invalidates the
|
||||
layer exactly when a cached result would stop being true, and it is documented
|
||||
with the other entrypoints in the README. `$GOPATH/bin` joins `PATH` because
|
||||
that is where bootstrap's `go install` lands and bootstrap verifies its
|
||||
installs against what `PATH` resolves — nothing in the image is shadowed by
|
||||
it, the directory does not exist until bootstrap runs. Everything added sits
|
||||
above `ARG CHECK_EPOCH`, and the `chown` and `USER builder` still precede
|
||||
`make check`. Verified: the guard fails the build with both versions named
|
||||
when the lint stage's linter is faked to a different version, and an
|
||||
unmodified build still passes it; bootstrap runs clean under Alpine's `sh` and
|
||||
its `apk` branch, installing `git` and `make` and finding the copied linter
|
||||
already at the pin; a second build served the bootstrap and dependency layers
|
||||
`CACHED` while both gates ran with a fresh epoch; a planted `unused` finding
|
||||
failed the build at the lint gate in 48.9s with the build stage's `make check`
|
||||
never starting; and the suite run in the image as `--user 0:0` fails
|
||||
`TestScanHardlinkRunFailsTogether`, so the drop to the unprivileged user is
|
||||
still load-bearing. That last check needs the Go test cache disabled — the
|
||||
first attempt reported `ok ... (cached)` as root, reusing the result the
|
||||
build-time run had left in the shared cache, which would have read as a pass.
|
||||
Build wall time, on a shared host running many concurrent builds and so noisy:
|
||||
2m13s on an unchanged tree, 2m17s and 4m29s for two builds after a source
|
||||
change, 5m14s cold. Only the cold one breaches the policy ceiling, and not
|
||||
because of this change — `chown -R builder:builder /src /home/builder` walks
|
||||
the module cache and re-runs on every source change, and it alone varied
|
||||
between 77s and 210s across those four builds, which is also the whole spread
|
||||
in the totals. The same cold measurement against `main` is 5m03s with a 209s
|
||||
`chown`. Filed as #43
|
||||
- bust the Docker layer cache for the gate steps, so `script/cibuild` and
|
||||
`script/docker` cannot report a green they did not earn (2026-08-09, branch
|
||||
`cibuild-cache-bust`, closes #32): both scripts were bare `docker build`
|
||||
invocations with no cache control, and the `Dockerfile` copies the tree before
|
||||
running its gates, so on an unchanged tree Docker served those layers from
|
||||
cache and the build exited 0 having executed nothing. That is not hypothetical
|
||||
here — every merge this repo has done is a non-fast-forward merge of an
|
||||
undiverged branch, so each merge commit's tree is byte-identical to the branch
|
||||
head's and each merge CI run was almost certainly a full cache hit; and PR
|
||||
#31's reviewer found `make docker` returning success as a 17-layer cache hit,
|
||||
catching it only by being suspicious. The fix is `ARG CHECK_EPOCH` with the
|
||||
scripts passing `--build-arg CHECK_EPOCH="$(date +%s)"`. Two details make or
|
||||
break it. `ARG` is scoped per stage and this `Dockerfile` has three gates
|
||||
across two — `make fmt-check` and `make lint` in the lint stage, `make check`
|
||||
in the build stage — so a single declaration would have left one stage
|
||||
silently cacheable; it is declared in both. And BuildKit hashes the expanded
|
||||
command, not the declaration, so a declared-but-unreferenced `ARG` invalidates
|
||||
nothing: each gate `RUN` echoes the epoch, which also puts the value in the
|
||||
build log as evidence the layer really ran. Placement is below the dependency
|
||||
layers on purpose — a build that goes cold every time would be a different
|
||||
bug, not a fix. Verified by running each script twice back to back on an
|
||||
unchanged tree under `BUILDKIT_PROGRESS=plain`: all three gates executed on
|
||||
all four runs, each with a fresh epoch in the log (`script/cibuild` 78.8s then
|
||||
61.1s; `script/docker` 61.1s then 53.4s), and twelve steps were still served
|
||||
`CACHED` in the steady state — both `go mod download`s, `apk add`, `adduser`,
|
||||
the `chown`, every `go.mod`/`go.sum` and source copy, the linter copy out of
|
||||
the lint stage, and the binary copy into the runtime stage. The lint stage
|
||||
still gates the build stage: with a deliberate `unused` finding planted in the
|
||||
tree, the build failed at `make lint` in 36.1s and the build-stage
|
||||
`make check` never started. The build stage also still drops to the
|
||||
unprivileged `builder` user before `make check`, which the suite depends on
|
||||
rather than merely prefers: forcing the same image to run the tests as root
|
||||
fails `TestScanHardlinkRunFailsTogether`, because root reads straight through
|
||||
the `chmod(0)` the test uses to prove hard links are read once. This is the
|
||||
local fix only; propagating it to the canonical templates is `prompts` #26
|
||||
- check the installed golangci-lint version in `script/bootstrap` instead of
|
||||
only its presence (2026-08-09, branch `bootstrap-version-check`, closes #24):
|
||||
`missing golangci-lint` meant any linter already on `PATH` satisfied the
|
||||
check, so the pin was never consulted and the v2.12.2 bump from #3 was inert
|
||||
on every host that already had one — this host ran v2.10.1 against a v2.12.2
|
||||
pin, `make check` went green, and `make docker` then rejected the same commit
|
||||
with findings the local gate never saw. The version now lives in one place,
|
||||
`GOLANGCI_LINT_VERSION`, with the `go install` module ref derived from it so a
|
||||
bump cannot half-apply; a `golangci_lint_version` helper parses
|
||||
`golangci-lint --version` (taking the field after the word `version` and
|
||||
tolerating an optional leading `v`, which the module ref carries and the
|
||||
binary's output does not), and any version that is not the pin — older, newer,
|
||||
absent or unparseable — is reinstalled. The install is then verified against
|
||||
the binary `PATH` actually resolves: `go install` writes into `GOBIN` (or
|
||||
`GOPATH/bin`) while `make lint` runs whichever `golangci-lint` comes first on
|
||||
`PATH`, so a wrong-version one sitting ahead of it — nix, apt, brew, apk, or
|
||||
the `/usr/local/bin` copy the `Dockerfile` builder stage makes — would swallow
|
||||
the install and leave the local gate disagreeing with CI under an affirmative
|
||||
`bootstrap complete`. Bootstrap now re-reads the effective version after
|
||||
installing and, on a mismatch, prints both paths and both versions to stderr
|
||||
and exits non-zero instead of claiming success; it does not reorder anyone's
|
||||
`PATH` or delete their binary. The `--version` call keeps its stderr
|
||||
connected, so a present-but-broken binary says why rather than reinstalling
|
||||
forever in silence, and is bounded by `timeout(1)` where that exists, so a
|
||||
wedged binary cannot hang bootstrap. `git`, `make` and `go` keep their
|
||||
presence-only checks and now say why in a comment: they are host
|
||||
package-manager tools the repo deliberately does not pin, with `go.mod`
|
||||
governing the language version and the digest-pinned images covering
|
||||
reproducible builds. Verified on this host by bootstrapping from v2.10.1 to
|
||||
v2.12.2 and running it again to a no-op, plus stub runs of the real script
|
||||
under `dash` covering a thirteen-input parse matrix (absent, older, newer,
|
||||
host-style, image-style, leading-`v`, stderr-only, empty, non-zero exit,
|
||||
impostor binary, `(devel)`, trailing `version`), a shadowed install that must
|
||||
exit non-zero, an install destination not on `PATH` at all, `GOBIN` set, and a
|
||||
wedged binary that must hit the timeout; `make check` and `make lint` are
|
||||
clean at v2.12.2, so v2.10.1 was not hiding any findings on `main`
|
||||
- unwind the hash worker pool on the error path (2026-08-09, branch
|
||||
`hash-pool-cleanup`, closes #6): `hashPhase` used to return the moment
|
||||
`recordRun` failed and abandon the pool — the feeder parked forever on a full
|
||||
`jobs` channel and every worker on a full `results` channel. That only stopped
|
||||
being invisible when #4 landed and `runScan` began unwinding instead of
|
||||
calling `os.Exit`. The pool is now an owned, context-aware `hashPool`: every
|
||||
blocking send in the feeder and the workers selects on `ctx.Done()`, `jobs` is
|
||||
closed on every path out, and `hashPhase` defers `pool.stop()`, which cancels
|
||||
and then drains `results` until the last goroutine has exited — draining is
|
||||
what frees a worker already parked on a send. `ctx` is threaded from
|
||||
`cmd.Context()` through `runScan`, `syncScan`, both worker pools and the whole
|
||||
database layer (it is the first parameter everywhere), so #5 can hand this
|
||||
path a signal and needs to add nothing else. The walk pool never leaked,
|
||||
because `walkPhase` always drains its events to close, but it has the same
|
||||
unbounded-send shape and #5 will give it an early return, so it gets the same
|
||||
treatment plus a `ctx.Err()` guard after the walk: a cancelled walk yields a
|
||||
partial size census, and every file it never reached looks vanished to the
|
||||
update phase. That phase's own `BeginTx` fails on the same cancelled context
|
||||
before deleting anything, so the guard is defence in depth rather than the
|
||||
only barrier — but it is the one that survives #5 deciding an interrupted scan
|
||||
may commit what it has. Tests drive `run(scan)` against a database whose
|
||||
insert trigger aborts, and assert both that the scan fails instead of hanging
|
||||
and that `runtime.NumGoroutine()` polls back to its pre-scan baseline; a
|
||||
second set cancels a scan part-way through the walk — deterministically, by
|
||||
counting the scan's own consultations of `ctx.Done()` rather than racing a
|
||||
timer — and asserts that it stops at the guard holding a partial census and a
|
||||
still-populated record index, with every record intact. The remaining
|
||||
cancellation branches of both pools are covered by direct tests of
|
||||
`sendEvent`, the walk workers, `dispatchDirs`, `feedHashJobs`, `hashWorker`
|
||||
and `hashPhase`
|
||||
- guarantee the database is closed on every fatal exit path (2026-08-09, branch
|
||||
`db-close-on-fatal`, closes #4): `fatalf` and its `os.Exit(1)` are gone, so
|
||||
the deferred `db.Close()` — and with it the SQLite WAL checkpoint — now
|
||||
actually runs when a subcommand fails; `runScan`, `runReport`, `runTrees`,
|
||||
`loadRecords` and `resolveRoots` return errors instead. The single exit point
|
||||
is `run` in `main.go`: it maps a `fatalError` (anything a subcommand returned)
|
||||
to exit 1 and cobra's own argument and flag errors to exit 2, which keeps a
|
||||
runtime failure from being reported as a usage error or printing the usage
|
||||
text. New `main_test.go` drives the CLI in-process and asserts the exit codes
|
||||
from README §Error handling plus the stdout/stderr split, including that a
|
||||
fatal error raised after the database is open leaves no `-wal`/`-shm` sidecar
|
||||
behind for `scan`, `report` or `trees`
|
||||
- update golangci-lint to v2.12.2 with the canonical config (2026-08-09, branch
|
||||
`golangci-v2.12.2`, merged as `38a01bd`, closes #3): bumped the pinned linter
|
||||
in the `Dockerfile` lint stage and `script/bootstrap` from v2.12.1 to v2.12.2,
|
||||
and replaced `.golangci.yml` with the canonical file — the linter settings
|
||||
(`lll`, `funlen`, `cyclop`, `dupl` thresholds) now live under
|
||||
`linters.settings` per the v2 schema, so they are actually applied; no new
|
||||
lint findings surfaced
|
||||
- convert Makefile targets to scripts-to-rule-them-all `script/` entrypoints
|
||||
like the other managed repos (2026-07-26, commit `3abeacf`, closes #1): all 12
|
||||
`script/` entrypoints exist (`bootstrap`, `setup`, `projectname`, `test`,
|
||||
`lint`, `fmt`, `fmt-check`, `check`, `docker`, `cibuild`, `precommit`,
|
||||
`install-precommit`) and every Makefile target is now a thin shim over them,
|
||||
matching the other managed repos
|
||||
- make the binary the default Make target (2026-07-24, branch
|
||||
`make-default-target`): plain `make` now builds `sfdupes` (previously it ran
|
||||
`check` plus `build`); `make build` remains as an alias
|
||||
- scan-wide phases, concurrent operands, batched updates (2026-07-24, branch
|
||||
`scan-wide-phases`): all operands seed the shared walk pool and every pass
|
||||
runs once over the whole scan, so totals and ETAs are scan-global; the
|
||||
per-operand walk/hash/update cycles and their stderr announcements are gone;
|
||||
the update pass commits in batched transactions — the filesystem is
|
||||
authoritative and the database an eventually-consistent reflection, so
|
||||
scan-level atomicity is not required
|
||||
- split the stat pass back out of the walk (2026-07-24, branch
|
||||
`parallel-phases`): phases are strictly sequential again — walk, stat, hash,
|
||||
update per operand — with parallelism only inside each phase; the walk
|
||||
enumerates paths with per-directory workers and the stat pass lstats them with
|
||||
per-file workers, restoring the exact total/ETA stat bar
|
||||
- announce each operand on stderr before its passes (2026-07-24, branch
|
||||
`scan-operand-progress`): with per-operand walk/hash/update cycles, a
|
||||
multi-operand run (e.g. `scan /srv/*`) showed pass totals that looked like the
|
||||
whole run's — an operator watching operand 3 of 14 hash 300k files concluded
|
||||
20M files were being skipped
|
||||
- parallel walk (2026-07-24, branch `parallel-walk`): the walk pass was a single
|
||||
goroutine and took hours at ~20M files on a busy pool (observed: 22M files in
|
||||
4h on a ZFS server); it is now a per-directory worker-pool traversal that
|
||||
records size/mtime during the walk (folding away the separate stat pass,
|
||||
halving metadata I/O), and each `PATH` operand commits in its own transaction
|
||||
so an interrupted scan keeps completed operands
|
||||
|
||||
- persistent scan database (2026-07-24, branch `persistent-database`): `scan`
|
||||
now maintains a SQLite database (`modernc.org/sqlite`, pure Go, cgo stays
|
||||
disabled) keyed by absolute path that survives between runs — a rescan hashes
|
||||
only new or changed files (by mtime/size), deletes records for files vanished
|
||||
from under the scanned operands, and leaves records outside them untouched, so
|
||||
`scan` can be cronned daily; `report` and `trees` read the database (no
|
||||
positional arguments) instead of a scan stream. Database at
|
||||
`/var/lib/sfdupes/db.sqlite`, overridable via `SFDUPES_DATABASE`; WAL
|
||||
journaling plus a single-transaction update keep a report run during a scan
|
||||
safe
|
||||
- add the `origin` remote (`git@git.eeqj.de:sneak/sfdupes.git`), tag `v0.0.1`,
|
||||
and push `main` plus tags (2026-07-23)
|
||||
- `scan` CLI rework (2026-07-23, branch `scan-required-paths`): required
|
||||
`PATH...` operands via cobra flags replacing the `/srv` `-root`
|
||||
default; new `-x`/`--one-file-system` flag (GNU convention) to stop
|
||||
at filesystem boundaries, which are crossed by default
|
||||
`PATH...` operands via cobra flags replacing the `/srv` `-root` default; new
|
||||
`-x`/`--one-file-system` flag (GNU convention) to stop at filesystem
|
||||
boundaries, which are crossed by default
|
||||
- bring the repo into full policy compliance (2026-07-23, branch
|
||||
`repo-policy-compliance`; checklist below)
|
||||
- `git init` with README-only first commit; code baseline committed on
|
||||
`main` (2026-07-22)
|
||||
- `git init` with README-only first commit; code baseline committed on `main`
|
||||
(2026-07-22)
|
||||
- implement `scan`, `report`, and `trees` subcommands (pre-git history)
|
||||
|
||||
# Future Steps
|
||||
|
||||
- convert Makefile targets to scripts-to-rule-them-all `script/`
|
||||
entrypoints like the other managed repos
|
||||
- tag `v0.0.1` once the compliance branch is merged
|
||||
- possible later features (explicitly out of scope per README):
|
||||
full-content verification of candidates, removal-script helpers
|
||||
- possible later features (explicitly out of scope per README): full-content
|
||||
verification of candidates, removal-script helpers
|
||||
|
||||
# Repo Policy Compliance
|
||||
|
||||
Audited 2026-07-22 against `REPO_POLICIES.md` (2026-07-06), the existing
|
||||
repo checklist, and the Go styleguide. Code is already gofmt-clean, so no
|
||||
standalone formatting commit is needed.
|
||||
Audited 2026-07-22 against `REPO_POLICIES.md` (2026-07-06), the existing repo
|
||||
checklist, and the Go styleguide. Code is already gofmt-clean, so no standalone
|
||||
formatting commit is needed.
|
||||
|
||||
- [x] `.gitignore` missing — the compiled `sfdupes` binary and
|
||||
`files.dat` sit untracked in the tree; needs OS/editor/Go
|
||||
artifacts plus secrets patterns
|
||||
- [x] `.gitignore` missing — the compiled `sfdupes` binary and `files.dat` sit
|
||||
untracked in the tree; needs OS/editor/Go artifacts plus secrets patterns
|
||||
- [x] `.editorconfig` missing
|
||||
- [x] `LICENSE` missing and README has no License section (MIT assumed
|
||||
from house convention — user to confirm)
|
||||
- [x] `LICENSE` missing and README has no License section (MIT assumed from
|
||||
house convention — user to confirm)
|
||||
- [x] `REPO_POLICIES.md` missing from repo root
|
||||
- [x] `.golangci.yml` missing (install canonical copy); code must then
|
||||
pass `make lint` (150 findings fixed; `make lint` is clean)
|
||||
- [x] `Makefile` lacks required targets `test`, `lint`, `fmt`,
|
||||
`fmt-check`, `docker`, `hooks`; `check` currently depends on
|
||||
`build`, which writes the binary (`make check` must not modify
|
||||
files)
|
||||
- [x] no tests — `go test ./...` has nothing to run; policy requires
|
||||
real tests with a 30-second timeout and the conditional `-v`
|
||||
rerun pattern (suite covers parsing, grouping, digests,
|
||||
suppression, hashing, and the scan pipeline; 64% coverage)
|
||||
- [x] `Dockerfile` missing — Go multistage with hash-pinned images:
|
||||
fail-fast lint stage, build stage running `make check`
|
||||
- [x] `.golangci.yml` missing (install canonical copy); code must then pass
|
||||
`make lint` (150 findings fixed; `make lint` is clean)
|
||||
- [x] `Makefile` lacks required targets `test`, `lint`, `fmt`, `fmt-check`,
|
||||
`docker`, `hooks`; `check` currently depends on `build`, which writes the
|
||||
binary (`make check` must not modify files)
|
||||
- [x] no tests — `go test ./...` has nothing to run; policy requires real tests
|
||||
with a 30-second timeout and the conditional `-v` rerun pattern (suite
|
||||
covers parsing, grouping, digests, suppression, hashing, and the scan
|
||||
pipeline; 64% coverage)
|
||||
- [x] `Dockerfile` missing — Go multistage with hash-pinned images: fail-fast
|
||||
lint stage, build stage running `make check`
|
||||
- [x] `.dockerignore` missing
|
||||
- [x] `.gitea/workflows/check.yml` missing (`docker build .` on push,
|
||||
checkout action pinned by commit SHA)
|
||||
- [x] `.gitea/workflows/check.yml` missing (`docker build .` on push, checkout
|
||||
action pinned by commit SHA)
|
||||
- [x] README lacks required sections: Description first line
|
||||
(name/purpose/category/license/author), Getting Started,
|
||||
Rationale, TODO, License, Author
|
||||
- [x] README non-goal "no git repository setup and no CI" is stale now
|
||||
that the repo is under git with CI
|
||||
- [x] pre-commit hook not installed (`make hooks` once the target
|
||||
exists)
|
||||
(name/purpose/category/license/author), Getting Started, Rationale, TODO,
|
||||
License, Author
|
||||
- [x] README non-goal "no git repository setup and no CI" is stale now that the
|
||||
repo is under git with CI
|
||||
- [x] pre-commit hook not installed (`make hooks` once the target exists)
|
||||
|
||||
Accepted divergences (no action):
|
||||
|
||||
- flat single-package layout with `.go` files in the repo root — fine
|
||||
for a small single-binary tool per the Go styleguide; the tracker
|
||||
audit agrees
|
||||
- `go test` runs without `-race` — the repo mandates `CGO_ENABLED=0`
|
||||
(pure-Go builds) and the race detector requires cgo
|
||||
- flat single-package layout with `.go` files in the repo root — fine for a
|
||||
small single-binary tool per the Go styleguide; the tracker audit agrees
|
||||
- `go test` runs without `-race` — the repo mandates `CGO_ENABLED=0` (pure-Go
|
||||
builds) and the race detector requires cgo
|
||||
|
||||
+793
@@ -0,0 +1,793 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/signal"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// This file gathers the tests for scan cancellation and worker-pool
|
||||
// unwinding. Everything it exercises lives in scan.go, so by the repo's
|
||||
// convention of one test file per source file it would belong in
|
||||
// scan_test.go. It is kept separate on purpose: cancellation behaviour
|
||||
// cuts across both the walk pool and the hash pool as a single concern,
|
||||
// and scan_test.go is already over 1,600 lines. That is the deliberate
|
||||
// exception the convention otherwise expects to be stated.
|
||||
|
||||
// poolUnwind bounds how long a test waits for a cancellation to take
|
||||
// effect: for a goroutine to return or a channel to close once its
|
||||
// context is cancelled, or for a signal to cancel the scan's context.
|
||||
// Only a failing run waits this long, and the bound is what makes that
|
||||
// failure an assertion instead of a hang. A call made without it, as
|
||||
// most of this file's scans are, has no bound: a regression that parks
|
||||
// it is caught only as the test binary's own timeout.
|
||||
const poolUnwind = 2 * time.Second
|
||||
|
||||
// walkClock is a context whose cancellation is driven by the scan's
|
||||
// own progress rather than by the wall clock: it cancels itself the
|
||||
// moment its Done method has been consulted n times. That is what
|
||||
// makes "cancel in the middle of the walk" reproducible instead of a
|
||||
// race against a timer.
|
||||
//
|
||||
// The accounting behind the n chosen by each test: every blocking
|
||||
// channel operation in the walk selects on Done, so the walk spends
|
||||
// one consultation per file event plus a couple per directory. The
|
||||
// index load that runs ahead of it also consults Done, but a bounded
|
||||
// number of times that does not grow with the record count. The tests
|
||||
// depend on that property, not on the bound's exact value: each test
|
||||
// sets n from the consultations of the walk, plus those of the hash
|
||||
// phase when it cancels mid-hash, far from both ends of the phase it
|
||||
// interrupts, so the cancellation lands inside that phase whatever the
|
||||
// record count.
|
||||
type walkClock struct {
|
||||
n int64
|
||||
seen atomic.Int64
|
||||
once sync.Once
|
||||
done chan struct{}
|
||||
}
|
||||
|
||||
// newWalkClock returns a context that cancels itself on the nth
|
||||
// consultation of its Done method.
|
||||
func newWalkClock(n int64) *walkClock {
|
||||
return &walkClock{n: n, done: make(chan struct{})}
|
||||
}
|
||||
|
||||
// Done returns the cancellation channel, cancelling the context on the
|
||||
// nth call and on every call after it. The same channel is returned
|
||||
// throughout, so a caller that took it before the cancellation still
|
||||
// observes the close.
|
||||
func (c *walkClock) Done() <-chan struct{} {
|
||||
if c.seen.Add(1) >= c.n {
|
||||
c.once.Do(func() { close(c.done) })
|
||||
}
|
||||
|
||||
return c.done
|
||||
}
|
||||
|
||||
// Err reports the cancellation without consuming a consultation, which
|
||||
// is what lets the post-walk guard read it without disturbing the
|
||||
// count.
|
||||
func (c *walkClock) Err() error {
|
||||
select {
|
||||
case <-c.done:
|
||||
return context.Canceled
|
||||
default:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
// Deadline reports no deadline: this context is cancelled by progress,
|
||||
// never by time.
|
||||
func (c *walkClock) Deadline() (time.Time, bool) {
|
||||
return time.Time{}, false
|
||||
}
|
||||
|
||||
// Value carries nothing.
|
||||
func (c *walkClock) Value(_ any) any {
|
||||
return nil
|
||||
}
|
||||
|
||||
// walkCancelDirs and walkCancelFilesPerDir shape the fixture for the
|
||||
// mid-walk cancellation test. Spreading the files over directories is
|
||||
// load-bearing: it is what bounds how much of the tree can still be
|
||||
// walked after the cancellation, since the workers drop every
|
||||
// directory still queued and only the handful already in flight can
|
||||
// emit anything more.
|
||||
const (
|
||||
walkCancelDirs = 100
|
||||
walkCancelFilesPerDir = 20
|
||||
walkCancelFiles = walkCancelDirs * walkCancelFilesPerDir
|
||||
walkCancelWorkers = 4
|
||||
// The most files the walkCancelWorkers directories already in
|
||||
// flight when the scan is cancelled can still emit, at
|
||||
// walkCancelFilesPerDir each. A file count, not a directory count.
|
||||
walkCancelInFlightFiles = walkCancelWorkers * walkCancelFilesPerDir
|
||||
)
|
||||
|
||||
// walkCancelAtDone is the consultation on which the fixture's context
|
||||
// cancels itself. A quarter of the file count is far past the index
|
||||
// load's fixed handful and far short of the walk's total, so the
|
||||
// cancellation lands deep inside the walk and nowhere near either end
|
||||
// of it.
|
||||
const walkCancelAtDone = walkCancelFiles / 4
|
||||
|
||||
// buildWalkCancelTree writes walkCancelFiles empty files spread over
|
||||
// walkCancelDirs subdirectories. Zero-length files are never opened by
|
||||
// the hasher, so the fixture costs directory entries and no read I/O
|
||||
// while still giving the walk thousands of events to emit.
|
||||
func buildWalkCancelTree(t *testing.T) string {
|
||||
t.Helper()
|
||||
|
||||
dir := t.TempDir()
|
||||
|
||||
for i := range walkCancelDirs {
|
||||
sub := filepath.Join(dir, "d"+strconv.Itoa(i))
|
||||
|
||||
err := os.Mkdir(sub, 0o750)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
writeEmptyFiles(t, sub, walkCancelFilesPerDir)
|
||||
}
|
||||
|
||||
return dir
|
||||
}
|
||||
|
||||
// assertRecordsIntact fails when the database no longer holds exactly
|
||||
// the records it held before, reporting the first difference rather
|
||||
// than dumping thousands of paths.
|
||||
func assertRecordsIntact(t *testing.T, db *sql.DB, before []string) {
|
||||
t.Helper()
|
||||
|
||||
got := recordPaths(dbRecords(t, db))
|
||||
if len(got) != len(before) {
|
||||
t.Fatalf("%d records after the cancelled scan, want %d",
|
||||
len(got), len(before))
|
||||
}
|
||||
|
||||
for i := range got {
|
||||
if got[i] != before[i] {
|
||||
t.Fatalf("record %d = %q after the cancelled scan, want %q",
|
||||
i, got[i], before[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestSyncScanCancelledMidWalkKeepsRecords is the regression net under
|
||||
// the post-walk guard. The scan is cancelled part-way through the
|
||||
// walk, so it reaches the guard holding a genuinely partial size
|
||||
// census and a still-populated index of records the walk never got to.
|
||||
// Every one of those records would look vanished to the update phase.
|
||||
// The guard is what stops the scan there, and this test is what
|
||||
// notices if it stops doing so: deleting the guard, or making it
|
||||
// unreachable, makes the scan carry its truncated view into the update
|
||||
// phase, which counts every record the walk never reached for removal.
|
||||
//
|
||||
// The syncScan call here is not bounded by poolUnwind: a regression
|
||||
// that left a worker pool parked would hang it, and that regression is
|
||||
// caught only by the test binary's own timeout, not by a quick
|
||||
// assertion.
|
||||
//
|
||||
//nolint:paralleltest // counts goroutines: must not run beside others
|
||||
func TestSyncScanCancelledMidWalkKeepsRecords(t *testing.T) {
|
||||
dir := buildWalkCancelTree(t)
|
||||
db := openTestDB(t)
|
||||
|
||||
st := syncTree(t, db, dir)
|
||||
if st.added != walkCancelFiles {
|
||||
t.Fatalf("setup scan added %d records, want %d",
|
||||
st.added, walkCancelFiles)
|
||||
}
|
||||
|
||||
before := recordPaths(dbRecords(t, db))
|
||||
base := baselineGoroutines(t)
|
||||
|
||||
st, err := syncScan(newWalkClock(walkCancelAtDone), db,
|
||||
[]string{dir}, walkCancelWorkers, false)
|
||||
|
||||
assertWalkGuardAborted(t, st, err)
|
||||
assertRecordsIntact(t, db, before)
|
||||
|
||||
if got := settledGoroutines(t, base); got > base {
|
||||
t.Errorf("goroutines = %d after the cancelled scan, want %d back",
|
||||
got, base)
|
||||
}
|
||||
}
|
||||
|
||||
// assertWalkGuardAborted checks that the scan stopped at the post-walk
|
||||
// guard: with a census that is neither empty (the walk really ran)
|
||||
// nor complete (it really was cut short), and with no record counted
|
||||
// for removal. A removal count means the partial census was carried
|
||||
// past the guard into the update phase, which is the failure this test
|
||||
// exists to catch.
|
||||
func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
|
||||
t.Helper()
|
||||
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("syncScan cancelled mid-walk = %v, want %v",
|
||||
err, context.Canceled)
|
||||
}
|
||||
|
||||
if st.unchanged == 0 {
|
||||
t.Fatalf("stats = %+v: the census is empty, so the walk never "+
|
||||
"ran and the guard was reached for the wrong reason", st)
|
||||
}
|
||||
|
||||
if st.unchanged >= walkCancelFiles {
|
||||
t.Fatalf("stats = %+v: the census covers the whole tree, so the "+
|
||||
"walk was not cut short", st)
|
||||
}
|
||||
|
||||
// The workers drop every directory still queued once the scan is
|
||||
// cancelled, so only the files in the directories already in flight
|
||||
// can add to the census after the fact. A census beyond that bound
|
||||
// would mean the cancellation was not observed where it should have
|
||||
// been.
|
||||
limit := walkCancelAtDone + walkCancelInFlightFiles
|
||||
if st.unchanged > limit {
|
||||
t.Errorf("census covers %d files, want at most %d: the walk kept "+
|
||||
"taking directories off the queue after cancellation",
|
||||
st.unchanged, limit)
|
||||
}
|
||||
|
||||
if st.removed != 0 {
|
||||
t.Errorf("stats = %+v: the scan counted records for removal from "+
|
||||
"a partial census", st)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSyncScanCancelledBeforeLoadIndex covers the trivial end of the
|
||||
// cancellation path: a scan handed a context that is already cancelled
|
||||
// fails in the index load, before the walk pool is ever started. It
|
||||
// says nothing about the post-walk guard — nothing downstream of
|
||||
// loadIndex runs at all — only that the failure surfaces as a
|
||||
// cancellation and that no record is touched on the way out.
|
||||
func TestSyncScanCancelledBeforeLoadIndex(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := buildSmokeTree(t)
|
||||
db := openTestDB(t)
|
||||
|
||||
syncTree(t, db, dir)
|
||||
|
||||
before := recordPaths(dbRecords(t, db))
|
||||
|
||||
ctx, cancel := context.WithCancel(t.Context())
|
||||
cancel()
|
||||
|
||||
st, err := syncScan(ctx, db, []string{dir}, walkCancelWorkers, false)
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("syncScan on a cancelled context = %v, want %v",
|
||||
err, context.Canceled)
|
||||
}
|
||||
|
||||
if st != (scanStats{}) {
|
||||
t.Errorf("stats = %+v, want none: the scan gave up in the index "+
|
||||
"load, before any phase ran", st)
|
||||
}
|
||||
|
||||
assertRecordsIntact(t, db, before)
|
||||
}
|
||||
|
||||
// hashCancelAtDone is the consultation on which the mid-hash test's
|
||||
// context cancels itself. The walk of buildWalkCancelTree spends about
|
||||
// one per file and three per directory, and the hash phase then one per
|
||||
// file hashed, so this lands about half way through the hash phase.
|
||||
const hashCancelAtDone = walkCancelFiles + 3*walkCancelDirs +
|
||||
walkCancelFiles/2
|
||||
|
||||
// TestSyncScanCancelledMidHashKeepsHashedRecords cancels a first scan
|
||||
// part-way through its hash phase. The fixture holds fewer files than a
|
||||
// batch, so every file hashed is still waiting to be committed: the scan
|
||||
// must commit them all before it returns, and the next scan must hash
|
||||
// only the rest.
|
||||
func TestSyncScanCancelledMidHashKeepsHashedRecords(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := buildWalkCancelTree(t)
|
||||
db := openTestDB(t)
|
||||
|
||||
st, err := syncScan(newWalkClock(hashCancelAtDone), db,
|
||||
[]string{dir}, walkCancelWorkers, false)
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("syncScan cancelled mid-hash = %v, want %v",
|
||||
err, context.Canceled)
|
||||
}
|
||||
|
||||
if st.walked != walkCancelFiles || st.added == 0 ||
|
||||
st.added >= walkCancelFiles {
|
||||
t.Fatalf("stats = %+v: want the walk complete and the hash phase "+
|
||||
"cut short", st)
|
||||
}
|
||||
|
||||
if got := len(dbRecords(t, db)); got != st.added {
|
||||
t.Errorf("%d records after the cancelled scan, want the %d it hashed",
|
||||
got, st.added)
|
||||
}
|
||||
|
||||
hashed := st.added
|
||||
|
||||
st = syncTree(t, db, dir)
|
||||
if st.added != walkCancelFiles-hashed || st.unchanged != hashed {
|
||||
t.Errorf("next scan stats = %+v, want %d added %d unchanged",
|
||||
st, walkCancelFiles-hashed, hashed)
|
||||
}
|
||||
}
|
||||
|
||||
// storedPaths opens the database at path as report does, which fails
|
||||
// unless it is a valid database, and returns its records' paths.
|
||||
func storedPaths(t *testing.T, path string) []string {
|
||||
t.Helper()
|
||||
|
||||
db, err := openReportDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
defer func() { _ = db.Close() }()
|
||||
|
||||
return recordPaths(dbRecords(t, db))
|
||||
}
|
||||
|
||||
// TestRunScanInterrupted calls the scan entrypoint with a context that
|
||||
// is already cancelled, as when a signal arrives at once. It must return
|
||||
// errInterrupted promptly with its one line on stderr, leave the
|
||||
// database valid and as it was, and leave nothing in the way of the
|
||||
// next scan, which must bring the database up to date.
|
||||
func TestRunScanInterrupted(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
stderr := captureStderr(t)
|
||||
dir := buildSmokeTree(t)
|
||||
|
||||
err := runScan(t.Context(), []string{dir}, walkCancelWorkers, false)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
before := storedPaths(t, path)
|
||||
|
||||
// A vanished file and a new one: the interrupted scan records
|
||||
// neither.
|
||||
gone := filepath.Join(dir, "a", "unique.bin")
|
||||
|
||||
err = os.Remove(gone)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
added := writeFile(t, dir, "a/new.bin", pattern(50, 10))
|
||||
shown := len(stderr())
|
||||
done := make(chan struct{})
|
||||
|
||||
go func() {
|
||||
defer close(done)
|
||||
|
||||
err = runScan(cancelledContext(t), []string{dir}, walkCancelWorkers,
|
||||
false)
|
||||
}()
|
||||
|
||||
awaitReturn(t, done, "runScan")
|
||||
|
||||
if !errors.Is(err, errInterrupted) {
|
||||
t.Fatalf("runScan on a cancelled context = %v, want %v",
|
||||
err, errInterrupted)
|
||||
}
|
||||
|
||||
want := "scan: interrupted after 0 files\n"
|
||||
if got := stderr()[shown:]; got != want {
|
||||
t.Errorf("stderr = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
assertNoSidecars(t, path)
|
||||
|
||||
if got := storedPaths(t, path); !slices.Equal(got, before) {
|
||||
t.Errorf("records = %q after the interrupted scan, want %q",
|
||||
got, before)
|
||||
}
|
||||
|
||||
err = runScan(t.Context(), []string{dir}, walkCancelWorkers, false)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
got := storedPaths(t, path)
|
||||
if slices.Contains(got, gone) || !slices.Contains(got, added) {
|
||||
t.Errorf("records = %q after the next scan, want %q gone and %q "+
|
||||
"added", got, gone, added)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRunScanInterruptedMidHash interrupts the scan entrypoint part-way
|
||||
// through its hash phase, after the database is open. It must return
|
||||
// errInterrupted, release the lock, end stderr with its line counting
|
||||
// every file the walk reached, close the database out of WAL mode, and
|
||||
// keep the records it hashed.
|
||||
func TestRunScanInterruptedMidHash(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
stderr := captureStderr(t)
|
||||
dir := buildWalkCancelTree(t)
|
||||
|
||||
err := runScan(newWalkClock(hashCancelAtDone), []string{dir},
|
||||
walkCancelWorkers, false)
|
||||
if !errors.Is(err, errInterrupted) {
|
||||
t.Fatalf("runScan interrupted mid-hash = %v, want %v",
|
||||
err, errInterrupted)
|
||||
}
|
||||
|
||||
holdScanLock(t, path)
|
||||
|
||||
want := fmt.Sprintf("scan: interrupted after %d files\n", walkCancelFiles)
|
||||
if got := stderr(); !strings.HasSuffix(got, want) {
|
||||
t.Errorf("stderr = %q, want it to end with %q", got, want)
|
||||
}
|
||||
|
||||
assertNoSidecars(t, path)
|
||||
|
||||
db, err := openReportDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
defer func() { _ = db.Close() }()
|
||||
|
||||
// A plain close also removes the sidecars, but leaves WAL mode on.
|
||||
var mode string
|
||||
|
||||
err = db.QueryRowContext(t.Context(), "PRAGMA journal_mode").Scan(&mode)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if mode != "delete" {
|
||||
t.Errorf("journal mode = %q after the interrupted scan, want %q",
|
||||
mode, "delete")
|
||||
}
|
||||
|
||||
kept := len(dbRecords(t, db))
|
||||
if kept == 0 || kept >= walkCancelFiles {
|
||||
t.Errorf("%d records after the interrupted scan, want those it "+
|
||||
"hashed: some but not all of the %d files", kept, walkCancelFiles)
|
||||
}
|
||||
}
|
||||
|
||||
// TestInterruptContextCatchesSIGTERM sends SIGTERM to the test process
|
||||
// while the scan's handler is installed, and checks that it cancels the
|
||||
// scan's context.
|
||||
//
|
||||
//nolint:paralleltest // signals the whole process: must not run beside a scan
|
||||
func TestInterruptContextCatchesSIGTERM(t *testing.T) {
|
||||
// Caught here as well, so that a handler that misses SIGTERM fails
|
||||
// this test instead of ending the test process.
|
||||
caught := make(chan os.Signal, 1)
|
||||
signal.Notify(caught, syscall.SIGTERM)
|
||||
|
||||
defer signal.Stop(caught)
|
||||
|
||||
ctx, stop := interruptContext(t.Context())
|
||||
defer stop()
|
||||
|
||||
err := syscall.Kill(os.Getpid(), syscall.SIGTERM)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
case <-time.After(poolUnwind):
|
||||
t.Fatal("SIGTERM did not cancel the scan's context")
|
||||
}
|
||||
}
|
||||
|
||||
// TestCommitFullBatchKeepsFailedBatch checks that a full batch whose
|
||||
// commit fails, as it does once the scan is interrupted, stays in the
|
||||
// batch, so that syncScan's final commit saves it.
|
||||
func TestCommitFullBatchKeepsFailedBatch(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
s := &scanState{db: openTestDB(t)}
|
||||
for i := range updateBatchSize {
|
||||
s.batch = append(s.batch, scanRec{path: "/f" + strconv.Itoa(i)})
|
||||
}
|
||||
|
||||
err := s.commitFullBatch(cancelledContext(t))
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("commitFullBatch on a cancelled context = %v, want %v",
|
||||
err, context.Canceled)
|
||||
}
|
||||
|
||||
if len(s.batch) != updateBatchSize {
|
||||
t.Errorf("batch holds %d records after the failed commit, want %d",
|
||||
len(s.batch), updateBatchSize)
|
||||
}
|
||||
}
|
||||
|
||||
// drainClosed counts the values received from ch until it closes,
|
||||
// failing the test if it does not close within poolUnwind. A pool that
|
||||
// ignored its cancellation leaves its channel open with its goroutines
|
||||
// parked, and this is what reports that as an assertion.
|
||||
func drainClosed[T any](t *testing.T, ch <-chan T, what string) int {
|
||||
t.Helper()
|
||||
|
||||
counted := make(chan int, 1)
|
||||
|
||||
go func() {
|
||||
n := 0
|
||||
for range ch {
|
||||
n++
|
||||
}
|
||||
|
||||
counted <- n
|
||||
}()
|
||||
|
||||
select {
|
||||
case n := <-counted:
|
||||
return n
|
||||
case <-time.After(poolUnwind):
|
||||
t.Fatalf("%s stayed open after cancellation", what)
|
||||
|
||||
return 0
|
||||
}
|
||||
}
|
||||
|
||||
// awaitReturn fails the test if done is not closed within poolUnwind.
|
||||
func awaitReturn(t *testing.T, done <-chan struct{}, what string) {
|
||||
t.Helper()
|
||||
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(poolUnwind):
|
||||
t.Fatalf("%s did not return after cancellation", what)
|
||||
}
|
||||
}
|
||||
|
||||
// cancelledContext returns a context that is already cancelled.
|
||||
func cancelledContext(t *testing.T) context.Context {
|
||||
t.Helper()
|
||||
|
||||
ctx, cancel := context.WithCancel(t.Context())
|
||||
cancel()
|
||||
|
||||
return ctx
|
||||
}
|
||||
|
||||
// TestSendEventAbandonsBlockedSend checks that a walk goroutine with an
|
||||
// event to deliver and nobody to deliver it to leaves on cancellation
|
||||
// instead of holding the pool open. The channel here is unbuffered and
|
||||
// unread, so the send can never complete.
|
||||
func TestSendEventAbandonsBlockedSend(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
done := make(chan struct{})
|
||||
events := make(chan walkEvent)
|
||||
|
||||
go func() {
|
||||
defer close(done)
|
||||
|
||||
sendEvent(cancelledContext(t), events, walkEvent{})
|
||||
}()
|
||||
|
||||
awaitReturn(t, done, "sendEvent")
|
||||
}
|
||||
|
||||
// TestWalkOneDirStopsWhenCancelled checks that a cancelled scan stops
|
||||
// reading a directory instead of going through the rest of its
|
||||
// entries. A walk that kept going would return the subdirectory below
|
||||
// to descend into. Unlike a file event, that return is not a send the
|
||||
// cancellation can abandon, so the test catches the regression every
|
||||
// time.
|
||||
func TestWalkOneDirStopsWhenCancelled(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
|
||||
err := os.Mkdir(filepath.Join(dir, "sub"), 0o750)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Unbuffered and unread: on a cancelled scan every send gives up.
|
||||
events := make(chan walkEvent)
|
||||
|
||||
subs := walkOneDir(cancelledContext(t), dirJob{path: dir}, false, events)
|
||||
if len(subs) != 0 {
|
||||
t.Errorf("cancelled walkOneDir returned %+v to descend into, "+
|
||||
"want none", subs)
|
||||
}
|
||||
}
|
||||
|
||||
// TestWalkWorkersDropQueuedDirs checks that cancelled walk workers keep
|
||||
// reading jobs and drop the directories rather than stopping their
|
||||
// read: the range over jobs has to run out for the pool to tear down
|
||||
// and close its event stream. The queued directory does not exist, so
|
||||
// a worker that walked it anyway would send a warning before
|
||||
// walkOneDir's own cancellation check could stop it. On a cancelled
|
||||
// scan that send delivers or gives up at random, so with 64 jobs
|
||||
// queued the regression has a one in 2^64 chance of passing.
|
||||
func TestWalkWorkersDropQueuedDirs(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
missing := filepath.Join(t.TempDir(), "missing")
|
||||
|
||||
jobs, _, events := startWalkWorkers(cancelledContext(t), 2, false)
|
||||
|
||||
for range 64 {
|
||||
jobs <- dirJob{path: missing}
|
||||
}
|
||||
|
||||
close(jobs)
|
||||
|
||||
if n := drainClosed(t, events, "the walk event stream"); n != 0 {
|
||||
t.Errorf("cancelled walk workers emitted %d events, want none", n)
|
||||
}
|
||||
}
|
||||
|
||||
// TestWalkWorkerAbandonsSubdirHandoff checks the other blocking send a
|
||||
// walk worker makes: handing discovered subdirectories back to the
|
||||
// dispatcher. Once the dispatcher has left, nothing drains that
|
||||
// channel, and a worker parked on it would hold the pool open forever.
|
||||
func TestWalkWorkerAbandonsSubdirHandoff(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
writeEmptyFiles(t, dir, 1)
|
||||
|
||||
err := os.Mkdir(filepath.Join(dir, "sub"), 0o750)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithCancel(t.Context())
|
||||
defer cancel()
|
||||
|
||||
jobs, subdirs, events := startWalkWorkers(ctx, 1, false)
|
||||
|
||||
// Fill the hand-back channel to its capacity — one slot per worker
|
||||
// — so the worker's own hand-back is certain to block.
|
||||
subdirs <- nil
|
||||
|
||||
jobs <- dirJob{path: dir}
|
||||
|
||||
// The file event proves the worker has read the directory and has
|
||||
// nothing left to do but the blocked hand-back.
|
||||
ev := <-events
|
||||
if ev.fail {
|
||||
t.Fatalf("walk event = %+v, want the fixture file", ev)
|
||||
}
|
||||
|
||||
cancel()
|
||||
close(jobs)
|
||||
drainClosed(t, events, "the walk event stream")
|
||||
}
|
||||
|
||||
// TestDispatchDirsClosesJobsWhenCancelled checks that a dispatcher
|
||||
// leaving on cancellation closes the job channel on its way out. The
|
||||
// workers range over that channel; a dispatcher that returned without
|
||||
// closing it would strand every one of them.
|
||||
func TestDispatchDirsClosesJobsWhenCancelled(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// Unbuffered and unread: with no worker pool behind it, the
|
||||
// dispatcher can only leave through its cancellation case.
|
||||
jobs := make(chan dirJob)
|
||||
subdirs := make(chan []dirJob)
|
||||
initial := []dirJob{{path: "/a"}, {path: "/b"}}
|
||||
|
||||
dispatchDirs(cancelledContext(t), initial, jobs, subdirs)
|
||||
|
||||
if n := drainClosed(t, jobs, "the walk job queue"); n > len(initial) {
|
||||
t.Errorf("dispatcher queued %d jobs, want at most %d",
|
||||
n, len(initial))
|
||||
}
|
||||
}
|
||||
|
||||
// TestFeedHashJobsClosesJobsWhenCancelled checks that the hash feeder
|
||||
// abandons the runs it has not queued yet and still closes the job
|
||||
// channel, which is what lets the workers' range terminate. The
|
||||
// receive on jobs below is not bounded: a feeder that returned without
|
||||
// closing jobs would leave that receive with no sender and no close, so
|
||||
// this regression is caught by the test binary's timeout rather than by
|
||||
// a bounded assertion.
|
||||
func TestFeedHashJobsClosesJobsWhenCancelled(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
done := make(chan struct{})
|
||||
// Unbuffered and unread until the feeder has returned, so the only
|
||||
// way out of the feeder is its cancellation case.
|
||||
jobs := make(chan []fileRec)
|
||||
runs := [][]fileRec{{{path: "a"}}, {{path: "b"}}}
|
||||
|
||||
go func() {
|
||||
defer close(done)
|
||||
|
||||
feedHashJobs(cancelledContext(t), runs, jobs)
|
||||
}()
|
||||
|
||||
awaitReturn(t, done, "feedHashJobs")
|
||||
|
||||
if _, ok := <-jobs; ok {
|
||||
t.Error("the hash job channel was left open after cancellation")
|
||||
}
|
||||
}
|
||||
|
||||
// TestHashWorkerDropsQueuedRuns checks that a cancelled hash worker
|
||||
// keeps reading jobs and drops the runs rather than reading files
|
||||
// nobody wants the hashes of — while still letting the range run out
|
||||
// so the pool tears down. The queued run names a file that does not
|
||||
// exist, so a worker that hashed it anyway would produce a result.
|
||||
//
|
||||
// hashWorker's other cancellation exit, abandoning the send of a
|
||||
// result, is reachable from the scan: stop cancels the pool before it
|
||||
// drains results, so a worker waiting on that send can leave through
|
||||
// it. The tests that stop a scan mid-hash, among them
|
||||
// TestScanHashWriteFailureUnwindsPool, reach it in some runs only,
|
||||
// depending on timing, and no test fails without it, since stop's
|
||||
// drain frees a waiting worker anyway. This test, for its part, catches
|
||||
// a removed drop check in some runs only: a worker that hashes the run
|
||||
// anyway then picks at random between sending the result and leaving.
|
||||
func TestHashWorkerDropsQueuedRuns(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
done := make(chan struct{})
|
||||
jobs := make(chan []fileRec, 1)
|
||||
results := make(chan hashResult, 1)
|
||||
|
||||
run := []fileRec{{path: filepath.Join(t.TempDir(), "missing"), size: 1}}
|
||||
|
||||
jobs <- run
|
||||
|
||||
close(jobs)
|
||||
|
||||
go func() {
|
||||
defer close(done)
|
||||
|
||||
hashWorker(cancelledContext(t), jobs, results, hashSignature)
|
||||
}()
|
||||
|
||||
awaitReturn(t, done, "hashWorker")
|
||||
|
||||
select {
|
||||
case r := <-results:
|
||||
t.Errorf("cancelled hash worker produced %+v, want the run dropped",
|
||||
r)
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
// TestHashPhaseCancelledReturnsContextError checks the result loop's
|
||||
// own exit: with the pool cancelled, no result will ever arrive, and
|
||||
// the loop must leave through the cancellation rather than wait for a
|
||||
// receive that cannot happen. This call is not bounded by poolUnwind: a
|
||||
// loop that dropped its cancellation case would block on that receive,
|
||||
// so the regression surfaces as the test binary's timeout rather than
|
||||
// as a bounded assertion.
|
||||
func TestHashPhaseCancelledReturnsContextError(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
s := &scanState{
|
||||
db: openTestDB(t),
|
||||
toHash: []fileRec{{path: "a", size: 1, dev: 1, ino: 1}},
|
||||
}
|
||||
|
||||
err := s.hashPhase(cancelledContext(t), 2)
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("hashPhase on a cancelled context = %v, want %v",
|
||||
err, context.Canceled)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,650 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io/fs"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strconv"
|
||||
|
||||
"golang.org/x/sys/unix"
|
||||
// The pure-Go SQLite driver, registered as "sqlite"; keeps cgo
|
||||
// disabled.
|
||||
_ "modernc.org/sqlite"
|
||||
)
|
||||
|
||||
// defaultDatabasePath is where the persistent scan database lives when
|
||||
// SFDUPES_DATABASE is not set.
|
||||
const defaultDatabasePath = "/var/lib/sfdupes/db.sqlite"
|
||||
|
||||
// databaseEnv is the environment variable that overrides the database
|
||||
// path.
|
||||
const databaseEnv = "SFDUPES_DATABASE"
|
||||
|
||||
// schemaVersion is the database schema version this build reads and
|
||||
// writes, stored in PRAGMA user_version.
|
||||
const schemaVersion = 1
|
||||
|
||||
// dbDirPerm is the mode for a database parent directory created by
|
||||
// scan.
|
||||
const dbDirPerm = 0o755
|
||||
|
||||
// lockFilePerm is the mode for the scan lock file. Anyone who can open
|
||||
// the file can hold the lock and keep every scan from running, so it
|
||||
// is open to its owner only.
|
||||
const lockFilePerm = 0o600
|
||||
|
||||
// createTableSQL is the schema applied to a fresh database. Paths are
|
||||
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8.
|
||||
const createTableSQL = `
|
||||
CREATE TABLE files (
|
||||
path BLOB PRIMARY KEY,
|
||||
size INTEGER NOT NULL,
|
||||
mtime INTEGER NOT NULL,
|
||||
head TEXT NOT NULL,
|
||||
tail TEXT NOT NULL,
|
||||
content TEXT NOT NULL
|
||||
) WITHOUT ROWID
|
||||
`
|
||||
|
||||
// createIndexSQL indexes the records by signature, so report can have
|
||||
// SQLite group them without sorting the whole table.
|
||||
const createIndexSQL = `
|
||||
CREATE INDEX files_signature ON files (size, head, tail, content)
|
||||
`
|
||||
|
||||
// upsertSQL inserts one file record, replacing any existing record for
|
||||
// the same path.
|
||||
const upsertSQL = `
|
||||
INSERT INTO files (path, size, mtime, head, tail, content)
|
||||
VALUES (?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT (path) DO UPDATE SET
|
||||
size = excluded.size, mtime = excluded.mtime,
|
||||
head = excluded.head, tail = excluded.tail,
|
||||
content = excluded.content
|
||||
`
|
||||
|
||||
// errNoDatabase reports a missing database file for report/trees.
|
||||
var errNoDatabase = errors.New(
|
||||
"no database (run \"sfdupes scan\" first, or set " + databaseEnv + ")")
|
||||
|
||||
// errSchemaVersion reports a database whose schema version this build
|
||||
// does not understand.
|
||||
var errSchemaVersion = errors.New("unsupported database schema version")
|
||||
|
||||
// errScanRunning reports that another scan holds the lock on the
|
||||
// database.
|
||||
var errScanRunning = errors.New("another scan is running")
|
||||
|
||||
// databasePath resolves the database location: SFDUPES_DATABASE when
|
||||
// set and non-empty, the compiled-in default otherwise.
|
||||
func databasePath() string {
|
||||
if p := os.Getenv(databaseEnv); p != "" {
|
||||
return p
|
||||
}
|
||||
|
||||
return defaultDatabasePath
|
||||
}
|
||||
|
||||
// scanParams are the connection parameters for scan: read-write, with
|
||||
// WAL journaling and a busy timeout, so a report can run while a cron
|
||||
// scan is in progress. closeScanDatabase leaves WAL mode again.
|
||||
const scanParams = "_pragma=busy_timeout(10000)" +
|
||||
"&_pragma=journal_mode(WAL)" +
|
||||
"&_pragma=synchronous(NORMAL)"
|
||||
|
||||
// reportParams are the connection parameters for report and trees:
|
||||
// read-only, with the same busy timeout. They set no journal mode,
|
||||
// because setting one is a write.
|
||||
const reportParams = "mode=ro" +
|
||||
"&_pragma=busy_timeout(10000)" +
|
||||
"&_pragma=query_only(1)"
|
||||
|
||||
// openDB opens the SQLite database at path with the connection
|
||||
// parameters params. It does not create or verify the schema.
|
||||
func openDB(path, params string) (*sql.DB, error) {
|
||||
db, err := sql.Open("sqlite", "file:"+path+"?"+params)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open database %s: %w", path, err)
|
||||
}
|
||||
|
||||
// A single connection avoids SQLITE_BUSY between this process's
|
||||
// own connections; concurrency lives in the worker pools, not in
|
||||
// parallel database access.
|
||||
db.SetMaxOpenConns(1)
|
||||
|
||||
return db, nil
|
||||
}
|
||||
|
||||
// lockScanDatabase takes the lock that keeps a second scan off the
|
||||
// database at path: an exclusive flock(2) on the file beside it named
|
||||
// path with ".lock" appended, created along with the database's parent
|
||||
// directory if missing. A lock held by another scan fails at once
|
||||
// instead of waiting. The lock lasts until the returned file is closed
|
||||
// or the process ends. The file is never deleted: a scan that deleted
|
||||
// it would let the next scan lock a new file while another still holds
|
||||
// the old one.
|
||||
func lockScanDatabase(path string) (*os.File, error) {
|
||||
err := os.MkdirAll(filepath.Dir(path), dbDirPerm)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("create database directory: %w", err)
|
||||
}
|
||||
|
||||
lockPath := path + ".lock"
|
||||
|
||||
//nolint:gosec // the operator chooses the database path
|
||||
f, err := os.OpenFile(lockPath, os.O_RDWR|os.O_CREATE, lockFilePerm)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
err = unix.Flock(int(f.Fd()), unix.LOCK_EX|unix.LOCK_NB)
|
||||
if err != nil {
|
||||
_ = f.Close()
|
||||
|
||||
if errors.Is(err, unix.EWOULDBLOCK) {
|
||||
return nil, fmt.Errorf("%w (lock held on %s)",
|
||||
errScanRunning, lockPath)
|
||||
}
|
||||
|
||||
return nil, fmt.Errorf("lock %s: %w", lockPath, err)
|
||||
}
|
||||
|
||||
return f, nil
|
||||
}
|
||||
|
||||
// openScanDatabase opens the database for the scan subcommand, creating
|
||||
// the file, its parent directory, and the schema as needed.
|
||||
func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
|
||||
err := os.MkdirAll(filepath.Dir(path), dbDirPerm)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("create database directory: %w", err)
|
||||
}
|
||||
|
||||
db, err := openDB(path, scanParams)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
err = initSchema(ctx, db)
|
||||
if err != nil {
|
||||
_ = db.Close()
|
||||
|
||||
return nil, fmt.Errorf("database %s: %w", path, err)
|
||||
}
|
||||
|
||||
return db, nil
|
||||
}
|
||||
|
||||
// closeScanDatabase switches the database at path from WAL back to
|
||||
// rollback-journal mode and closes it. Out of WAL mode the database
|
||||
// file alone holds the whole database, so a reader needs no -wal or
|
||||
// -shm file beside it, nor write access to create them. The switch
|
||||
// fails while a report has the database open; the database then stays
|
||||
// in WAL mode, still readable, until a later scan closes it.
|
||||
func closeScanDatabase(ctx context.Context, db *sql.DB, path string) {
|
||||
// Runs on the way out of a cancelled scan too.
|
||||
_, err := db.ExecContext(context.WithoutCancel(ctx),
|
||||
"PRAGMA journal_mode = DELETE")
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "scan: database %s left in WAL mode: %v\n",
|
||||
path, err)
|
||||
}
|
||||
|
||||
_ = db.Close()
|
||||
}
|
||||
|
||||
// openReportDatabase opens an existing database for the report and
|
||||
// trees subcommands. A missing database file is an error directing the
|
||||
// user to run scan first; the schema version must match exactly.
|
||||
func openReportDatabase(ctx context.Context,
|
||||
path string,
|
||||
) (*sql.DB, error) {
|
||||
_, err := os.Stat(path)
|
||||
if errors.Is(err, fs.ErrNotExist) {
|
||||
return nil, fmt.Errorf("%s: %w", path, errNoDatabase)
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("database: %w", err)
|
||||
}
|
||||
|
||||
db, err := openDB(path, reportParams)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
v, err := userVersion(ctx, db)
|
||||
if err == nil && v == 0 {
|
||||
// An empty database passes this check and fails the version
|
||||
// check below.
|
||||
err = checkUnversioned(ctx, db)
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
_ = db.Close()
|
||||
|
||||
return nil, fmt.Errorf("database %s: %w", path, err)
|
||||
}
|
||||
|
||||
if v != schemaVersion {
|
||||
_ = db.Close()
|
||||
|
||||
return nil, fmt.Errorf("database %s: version %d, want %d: %w",
|
||||
path, v, schemaVersion, errSchemaVersion)
|
||||
}
|
||||
|
||||
return db, nil
|
||||
}
|
||||
|
||||
// initSchema creates the schema on a fresh database and verifies the
|
||||
// schema version on an existing one.
|
||||
func initSchema(ctx context.Context, db *sql.DB) error {
|
||||
v, err := userVersion(ctx, db)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
switch v {
|
||||
case 0:
|
||||
err = checkUnversioned(ctx, db)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
return createSchema(ctx, db)
|
||||
case schemaVersion:
|
||||
return nil
|
||||
default:
|
||||
return fmt.Errorf("version %d, want %d: %w",
|
||||
v, schemaVersion, errSchemaVersion)
|
||||
}
|
||||
}
|
||||
|
||||
// checkUnversioned checks a database at user_version 0 before it is
|
||||
// taken for an empty one. createSchema creates the files table and
|
||||
// sets the version together, so a files table at version 0 was made by
|
||||
// something else. Adopting it could corrupt unrelated data, so that is
|
||||
// a schema-version error telling the operator to remove the file and
|
||||
// rescan.
|
||||
func checkUnversioned(ctx context.Context, db *sql.DB) error {
|
||||
var name string
|
||||
|
||||
err := db.QueryRowContext(ctx,
|
||||
"SELECT name FROM sqlite_master "+
|
||||
"WHERE type = 'table' AND name = 'files'").Scan(&name)
|
||||
|
||||
switch {
|
||||
case err == nil:
|
||||
return fmt.Errorf(
|
||||
"has a files table but no schema version; "+
|
||||
"remove the file and rescan: %w", errSchemaVersion)
|
||||
case errors.Is(err, sql.ErrNoRows):
|
||||
return nil
|
||||
default:
|
||||
return fmt.Errorf("check for files table: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
// createSchema applies the schema to a fresh database and stamps the
|
||||
// schema version in one transaction, so a creation stopped partway, by
|
||||
// an interrupt or an error, leaves an empty database the next scan
|
||||
// sets up, never a files table at version 0, which checkUnversioned
|
||||
// refuses.
|
||||
func createSchema(ctx context.Context, db *sql.DB) error {
|
||||
tx, err := db.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create schema: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = tx.Rollback() }()
|
||||
|
||||
_, err = tx.ExecContext(ctx, createTableSQL)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create schema: %w", err)
|
||||
}
|
||||
|
||||
_, err = tx.ExecContext(ctx, createIndexSQL)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create schema: %w", err)
|
||||
}
|
||||
|
||||
_, err = tx.ExecContext(ctx,
|
||||
"PRAGMA user_version = "+strconv.Itoa(schemaVersion))
|
||||
if err != nil {
|
||||
return fmt.Errorf("set schema version: %w", err)
|
||||
}
|
||||
|
||||
err = tx.Commit()
|
||||
if err != nil {
|
||||
return fmt.Errorf("create schema: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// userVersion reads the database's PRAGMA user_version.
|
||||
func userVersion(ctx context.Context, db *sql.DB) (int, error) {
|
||||
var v int
|
||||
|
||||
err := db.QueryRowContext(ctx, "PRAGMA user_version").Scan(&v)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("read schema version: %w", err)
|
||||
}
|
||||
|
||||
return v, nil
|
||||
}
|
||||
|
||||
// loadFileRows streams every record to fn in path order: byte order,
|
||||
// which is the order of the primary key, so SQLite does not sort.
|
||||
func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
|
||||
rows, err := db.QueryContext(ctx,
|
||||
"SELECT path, size, mtime, head, tail, content FROM files "+
|
||||
"ORDER BY path")
|
||||
if err != nil {
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
for rows.Next() {
|
||||
var (
|
||||
path []byte
|
||||
r scanRec
|
||||
)
|
||||
|
||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
|
||||
&r.content)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read record: %w", err)
|
||||
}
|
||||
|
||||
r.path = string(path)
|
||||
fn(r)
|
||||
}
|
||||
|
||||
err = rows.Err()
|
||||
if err != nil {
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// dupeRowsSQL selects every record in a duplicate group, with the
|
||||
// group's first path. A group is the records with a content hash that
|
||||
// share a size, head, tail, and content, when there are two or more of
|
||||
// them. The rows come in report order: groups by size descending, then
|
||||
// by first path, and each group's paths ascending.
|
||||
const dupeRowsSQL = `
|
||||
SELECT g.first, f.path, f.size
|
||||
FROM files AS f
|
||||
JOIN (
|
||||
SELECT size, head, tail, content, MIN(path) AS first
|
||||
FROM files
|
||||
WHERE content <> ''
|
||||
GROUP BY size, head, tail, content
|
||||
HAVING COUNT(*) > 1
|
||||
) AS g USING (size, head, tail, content)
|
||||
ORDER BY f.size DESC, g.first, f.path
|
||||
`
|
||||
|
||||
// loadDupeRows streams the rows of dupeRowsSQL to fn and returns the
|
||||
// number of records in the database. The count and the rows are read
|
||||
// in one transaction, so they agree while a scan is committing. An
|
||||
// error from fn stops the reading and is returned as it is.
|
||||
func loadDupeRows(ctx context.Context, db *sql.DB,
|
||||
fn func(first, path string, size int64) error,
|
||||
) (int, error) {
|
||||
// Everything goes through tx: the report connection is the only
|
||||
// one, so a query on db would wait for tx forever.
|
||||
tx, err := db.BeginTx(ctx, &sql.TxOptions{ReadOnly: true})
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = tx.Rollback() }()
|
||||
|
||||
var records int
|
||||
|
||||
err = tx.QueryRowContext(ctx, "SELECT COUNT(*) FROM files").Scan(&records)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
rows, err := tx.QueryContext(ctx, dupeRowsSQL)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
for rows.Next() {
|
||||
var (
|
||||
first, path []byte
|
||||
size int64
|
||||
)
|
||||
|
||||
err = rows.Scan(&first, &path, &size)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("read record: %w", err)
|
||||
}
|
||||
|
||||
err = fn(string(first), string(path), size)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
}
|
||||
|
||||
err = rows.Err()
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
return records, nil
|
||||
}
|
||||
|
||||
// loadFileMeta streams every record's path, size, mtime, and whether
|
||||
// it carries hashes to fn. Scan change detection needs no hash
|
||||
// values, and skipping the hash columns keeps the scan's in-memory
|
||||
// index small on multi-million-file databases.
|
||||
func loadFileMeta(ctx context.Context, db *sql.DB,
|
||||
fn func(path string, size, mtime int64, hashed bool),
|
||||
) error {
|
||||
rows, err := db.QueryContext(ctx,
|
||||
"SELECT path, size, mtime, head <> '' FROM files")
|
||||
if err != nil {
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
for rows.Next() {
|
||||
var (
|
||||
path []byte
|
||||
size, mtime int64
|
||||
hashed int64
|
||||
)
|
||||
|
||||
err = rows.Scan(&path, &size, &mtime, &hashed)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read record: %w", err)
|
||||
}
|
||||
|
||||
fn(string(path), size, mtime, hashed != 0)
|
||||
}
|
||||
|
||||
err = rows.Err()
|
||||
if err != nil {
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// contentCandidatesSQL selects every record of at least headTailMin
|
||||
// bytes whose size, head, and tail equal another record's, in each
|
||||
// group (the records sharing a size, head, and tail) where at least one
|
||||
// record has no content hash, with whether each record has one. SQLite
|
||||
// does the grouping, so no other record's hashes are loaded into
|
||||
// memory; the rows come ordered by size, head, and tail, so each
|
||||
// group's rows arrive together.
|
||||
const contentCandidatesSQL = `
|
||||
SELECT f.path, f.size, f.mtime, f.head, f.tail, f.content <> ''
|
||||
FROM files AS f
|
||||
JOIN (
|
||||
SELECT size, head, tail
|
||||
FROM files
|
||||
WHERE size >= ? AND head <> ''
|
||||
GROUP BY size, head, tail
|
||||
HAVING COUNT(*) > 1 AND SUM(content = '') > 0
|
||||
) AS g USING (size, head, tail)
|
||||
ORDER BY size, head, tail
|
||||
`
|
||||
|
||||
// loadContentCandidates streams the rows of contentCandidatesSQL to fn:
|
||||
// each record, without its content hash, and whether it has one.
|
||||
func loadContentCandidates(ctx context.Context, db *sql.DB,
|
||||
fn func(r scanRec, hashed bool),
|
||||
) error {
|
||||
rows, err := db.QueryContext(ctx, contentCandidatesSQL, headTailMin)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = rows.Close() }()
|
||||
|
||||
for rows.Next() {
|
||||
var (
|
||||
path []byte
|
||||
r scanRec
|
||||
hashed int64
|
||||
)
|
||||
|
||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, &hashed)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read record: %w", err)
|
||||
}
|
||||
|
||||
r.path = string(path)
|
||||
fn(r, hashed != 0)
|
||||
}
|
||||
|
||||
err = rows.Err()
|
||||
if err != nil {
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// updateBatchSize is the number of record changes committed per
|
||||
// transaction during the update pass. The filesystem is authoritative
|
||||
// and the database an eventually-consistent reflection of it, so
|
||||
// scan-level atomicity is not required; smaller transactions keep the
|
||||
// WAL small and let concurrent reports observe progress.
|
||||
const updateBatchSize = 10000
|
||||
|
||||
// applyChanges writes one scan's database changes — upserts for new and
|
||||
// changed files, deletes for vanished ones — in batched transactions.
|
||||
// Progress is rendered on prog (one increment per change).
|
||||
func applyChanges(ctx context.Context, db *sql.DB, upserts []scanRec,
|
||||
deletes []string, prog *progress,
|
||||
) error {
|
||||
for batch := range slices.Chunk(upserts, updateBatchSize) {
|
||||
err := applyBatch(ctx, db, batch, nil, prog)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
for batch := range slices.Chunk(deletes, updateBatchSize) {
|
||||
err := applyBatch(ctx, db, nil, batch, prog)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// applyBatch commits one batch of upserts and deletes in a single
|
||||
// transaction.
|
||||
func applyBatch(ctx context.Context, db *sql.DB, upserts []scanRec,
|
||||
deletes []string, prog *progress,
|
||||
) error {
|
||||
tx, err := db.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return fmt.Errorf("begin transaction: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = tx.Rollback() }()
|
||||
|
||||
err = execUpserts(ctx, tx, upserts, prog)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
err = execDeletes(ctx, tx, deletes, prog)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
err = tx.Commit()
|
||||
if err != nil {
|
||||
return fmt.Errorf("commit: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// execUpserts inserts or updates one record per new or changed file.
|
||||
func execUpserts(ctx context.Context, tx *sql.Tx, upserts []scanRec,
|
||||
prog *progress,
|
||||
) error {
|
||||
st, err := tx.PrepareContext(ctx, upsertSQL)
|
||||
if err != nil {
|
||||
return fmt.Errorf("prepare upsert: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = st.Close() }()
|
||||
|
||||
for _, r := range upserts {
|
||||
_, err = st.ExecContext(ctx,
|
||||
[]byte(r.path), r.size, r.mtime, r.head, r.tail, r.content)
|
||||
if err != nil {
|
||||
return fmt.Errorf("upsert %s: %w", r.path, err)
|
||||
}
|
||||
|
||||
prog.increment()
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// execDeletes removes the records for paths no longer present.
|
||||
func execDeletes(ctx context.Context, tx *sql.Tx, deletes []string,
|
||||
prog *progress,
|
||||
) error {
|
||||
st, err := tx.PrepareContext(ctx, "DELETE FROM files WHERE path = ?")
|
||||
if err != nil {
|
||||
return fmt.Errorf("prepare delete: %w", err)
|
||||
}
|
||||
|
||||
defer func() { _ = st.Close() }()
|
||||
|
||||
for _, p := range deletes {
|
||||
_, err = st.ExecContext(ctx, []byte(p))
|
||||
if err != nil {
|
||||
return fmt.Errorf("delete %s: %w", p, err)
|
||||
}
|
||||
|
||||
prog.increment()
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
+340
@@ -0,0 +1,340 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// testDBPath returns a database path inside a fresh temp dir.
|
||||
func testDBPath(t *testing.T) string {
|
||||
t.Helper()
|
||||
|
||||
return filepath.Join(t.TempDir(), "db.sqlite")
|
||||
}
|
||||
|
||||
// openTestDB creates a fresh scan database in a temp dir.
|
||||
func openTestDB(t *testing.T) *sql.DB {
|
||||
t.Helper()
|
||||
|
||||
db, err := openScanDatabase(t.Context(), testDBPath(t))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
t.Cleanup(func() { _ = db.Close() })
|
||||
|
||||
return db
|
||||
}
|
||||
|
||||
func TestDatabasePath(t *testing.T) {
|
||||
t.Setenv(databaseEnv, "")
|
||||
|
||||
if got := databasePath(); got != defaultDatabasePath {
|
||||
t.Errorf("databasePath() = %q, want %q", got, defaultDatabasePath)
|
||||
}
|
||||
|
||||
t.Setenv(databaseEnv, "/custom/place.sqlite")
|
||||
|
||||
if got := databasePath(); got != "/custom/place.sqlite" {
|
||||
t.Errorf("databasePath() = %q, want the env override", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOpenScanDatabaseCreates(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// The parent directory does not exist yet; scan must create it.
|
||||
path := filepath.Join(t.TempDir(), "nested", "dir", "db.sqlite")
|
||||
|
||||
db, err := openScanDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatalf("openScanDatabase: %v", err)
|
||||
}
|
||||
|
||||
v, err := userVersion(t.Context(), db)
|
||||
if err != nil || v != schemaVersion {
|
||||
t.Fatalf("userVersion = %d, %v; want %d, nil", v, err, schemaVersion)
|
||||
}
|
||||
|
||||
_ = db.Close()
|
||||
|
||||
// Reopening an existing database must succeed and find the schema.
|
||||
db, err = openScanDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatalf("reopen: %v", err)
|
||||
}
|
||||
|
||||
defer func() { _ = db.Close() }()
|
||||
|
||||
if recs := dbRecords(t, db); len(recs) != 0 {
|
||||
t.Fatalf("records = %v, want none", recs)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOpenDatabaseUnversionedForeign(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// A database that has a files table but user_version 0, written by
|
||||
// some other tool. report, trees and scan must refuse it with the
|
||||
// schema-version error, not adopt it and not emit a raw SQLite
|
||||
// "table files already exists".
|
||||
path := testDBPath(t)
|
||||
|
||||
db, err := sql.Open("sqlite", path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
_, err = db.ExecContext(t.Context(), "CREATE TABLE files (x INTEGER)")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
_ = db.Close()
|
||||
|
||||
_, err = openReportDatabase(t.Context(), path)
|
||||
if !errors.Is(err, errSchemaVersion) ||
|
||||
!strings.Contains(err.Error(), "remove the file and rescan") {
|
||||
t.Fatalf("report: err = %v, want errSchemaVersion telling the "+
|
||||
"operator to remove the file and rescan", err)
|
||||
}
|
||||
|
||||
_, err = openScanDatabase(t.Context(), path)
|
||||
if !errors.Is(err, errSchemaVersion) ||
|
||||
!strings.Contains(err.Error(), "remove the file and rescan") {
|
||||
t.Fatalf("scan: err = %v, want errSchemaVersion telling the "+
|
||||
"operator to remove the file and rescan", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSchemaCreationStoppedPartway(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// A first scan stopped while creating the schema must leave a
|
||||
// database the next scan accepts. max_page_count(2) leaves room for
|
||||
// the files table but not its index, so schema creation fails right
|
||||
// after CREATE TABLE, a point an interrupt could also stop it at.
|
||||
path := testDBPath(t)
|
||||
|
||||
db, err := openDB(path, scanParams+"&_pragma=max_page_count(2)")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = initSchema(t.Context(), db)
|
||||
_ = db.Close()
|
||||
|
||||
if err == nil {
|
||||
t.Fatal("initSchema with no room for the index succeeded")
|
||||
}
|
||||
|
||||
db, err = openScanDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatalf("next scan: %v", err)
|
||||
}
|
||||
|
||||
defer func() { _ = db.Close() }()
|
||||
|
||||
v, err := userVersion(t.Context(), db)
|
||||
if err != nil || v != schemaVersion {
|
||||
t.Fatalf("userVersion = %d, %v; want %d, nil", v, err, schemaVersion)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOpenReportDatabaseMissing(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
_, err := openReportDatabase(t.Context(), testDBPath(t))
|
||||
if !errors.Is(err, errNoDatabase) {
|
||||
t.Fatalf("err = %v, want errNoDatabase", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOpenDatabaseVersionMismatch(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// A database stamped with a schema version other than 0 and
|
||||
// schemaVersion. report, trees and scan must all refuse it.
|
||||
path := testDBPath(t)
|
||||
|
||||
db, err := openScanDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
_, err = db.ExecContext(context.Background(), "PRAGMA user_version = 99")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
_ = db.Close()
|
||||
|
||||
_, err = openReportDatabase(t.Context(), path)
|
||||
if !errors.Is(err, errSchemaVersion) {
|
||||
t.Fatalf("report: err = %v, want errSchemaVersion", err)
|
||||
}
|
||||
|
||||
_, err = openScanDatabase(t.Context(), path)
|
||||
if !errors.Is(err, errSchemaVersion) {
|
||||
t.Fatalf("scan: err = %v, want errSchemaVersion", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOpenReportDatabaseOK(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
path := testDBPath(t)
|
||||
|
||||
db, err := openScanDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
_ = db.Close()
|
||||
|
||||
db, err = openReportDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatalf("openReportDatabase: %v", err)
|
||||
}
|
||||
|
||||
_ = db.Close()
|
||||
}
|
||||
|
||||
func TestCloseScanDatabaseWhileReportOpen(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// A report holding the database open stops scan from taking it out
|
||||
// of WAL mode. The -wal and -shm files must then stay beside it, so
|
||||
// that a later report still needs only read access.
|
||||
path := testDBPath(t)
|
||||
|
||||
scanDB, err := openScanDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
reportDB, err := openReportDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
closeScanDatabase(t.Context(), scanDB, path)
|
||||
|
||||
_ = reportDB.Close()
|
||||
|
||||
_, err = os.Stat(path + "-wal")
|
||||
if err != nil {
|
||||
t.Fatalf("no -wal left: the switch out of WAL mode was not "+
|
||||
"stopped: %v", err)
|
||||
}
|
||||
|
||||
makeReadOnly(t, path)
|
||||
|
||||
reportDB, err = openReportDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatalf("openReportDatabase: %v", err)
|
||||
}
|
||||
|
||||
defer func() { _ = reportDB.Close() }()
|
||||
|
||||
err = loadFileRows(t.Context(), reportDB, func(scanRec) {})
|
||||
if err != nil {
|
||||
t.Fatalf("loadFileRows: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyChangesRoundTrip(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := openTestDB(t)
|
||||
|
||||
// Paths may contain tabs and newlines; the database must store
|
||||
// them byte-exactly. Every hash, content included, comes back as
|
||||
// written.
|
||||
recs := []scanRec{
|
||||
{
|
||||
size: 2, mtime: 20, head: "h2", tail: "t2", content: "c2",
|
||||
path: "/a/tab\tnew\nline",
|
||||
},
|
||||
{size: 1, mtime: 10, head: "h1", tail: "t1", content: "c1", path: "/a/x"},
|
||||
}
|
||||
|
||||
err := applyChanges(t.Context(), db, recs, nil,
|
||||
newProgress("update", 2))
|
||||
if err != nil {
|
||||
t.Fatalf("applyChanges: %v", err)
|
||||
}
|
||||
|
||||
// The records come back in path order, which is the order of recs.
|
||||
got := dbRecords(t, db)
|
||||
if !slices.Equal(got, recs) {
|
||||
t.Fatalf("rows = %+v, want %+v", got, recs)
|
||||
}
|
||||
|
||||
// An upsert for an existing path updates in place; a delete
|
||||
// removes exactly its path.
|
||||
upd := scanRec{
|
||||
size: 3, mtime: 30, head: "h3", tail: "t3", content: "c3", path: "/a/x",
|
||||
}
|
||||
|
||||
err = applyChanges(t.Context(), db, []scanRec{upd},
|
||||
[]string{"/a/tab\tnew\nline"}, newProgress("update", 2))
|
||||
if err != nil {
|
||||
t.Fatalf("applyChanges: %v", err)
|
||||
}
|
||||
|
||||
got = dbRecords(t, db)
|
||||
if len(got) != 1 || got[0] != upd {
|
||||
t.Fatalf("rows = %+v, want just %+v", got, upd)
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyChangesBatching(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := openTestDB(t)
|
||||
|
||||
// One more change than the batch size, so the update spans two
|
||||
// transactions.
|
||||
n := updateBatchSize + 1
|
||||
|
||||
recs := make([]scanRec, 0, n)
|
||||
for i := range n {
|
||||
recs = append(recs, scanRec{
|
||||
size: int64(i), mtime: 1, head: "h", tail: "t",
|
||||
path: fmt.Sprintf("/batch/%07d", i),
|
||||
})
|
||||
}
|
||||
|
||||
err := applyChanges(t.Context(), db, recs, nil,
|
||||
newProgress("update", int64(n)))
|
||||
if err != nil {
|
||||
t.Fatalf("applyChanges: %v", err)
|
||||
}
|
||||
|
||||
if got := dbRecords(t, db); len(got) != n {
|
||||
t.Fatalf("records = %d, want %d", len(got), n)
|
||||
}
|
||||
|
||||
deletes := make([]string, 0, n)
|
||||
for _, r := range recs {
|
||||
deletes = append(deletes, r.path)
|
||||
}
|
||||
|
||||
err = applyChanges(t.Context(), db, nil, deletes,
|
||||
newProgress("update", int64(n)))
|
||||
if err != nil {
|
||||
t.Fatalf("applyChanges deletes: %v", err)
|
||||
}
|
||||
|
||||
if got := dbRecords(t, db); len(got) != 0 {
|
||||
t.Fatalf("records = %d, want 0", len(got))
|
||||
}
|
||||
}
|
||||
@@ -5,13 +5,22 @@ go 1.25.7
|
||||
require (
|
||||
github.com/schollz/progressbar/v3 v3.19.1
|
||||
github.com/spf13/cobra v1.10.2
|
||||
golang.org/x/sys v0.46.0
|
||||
golang.org/x/term v0.44.0
|
||||
modernc.org/sqlite v1.54.0
|
||||
)
|
||||
|
||||
require (
|
||||
github.com/dustin/go-humanize v1.0.1 // indirect
|
||||
github.com/google/uuid v1.6.0 // indirect
|
||||
github.com/inconshreveable/mousetrap v1.1.0 // indirect
|
||||
github.com/mattn/go-isatty v0.0.22 // indirect
|
||||
github.com/mitchellh/colorstring v0.0.0-20190213212951-d06e56a500db // indirect
|
||||
github.com/ncruces/go-strftime v1.0.0 // indirect
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
||||
github.com/rivo/uniseg v0.4.7 // indirect
|
||||
github.com/spf13/pflag v1.0.9 // indirect
|
||||
golang.org/x/sys v0.46.0 // indirect
|
||||
golang.org/x/term v0.44.0 // indirect
|
||||
modernc.org/libc v1.74.1 // indirect
|
||||
modernc.org/mathutil v1.7.1 // indirect
|
||||
modernc.org/memory v1.11.0 // indirect
|
||||
)
|
||||
|
||||
@@ -3,14 +3,28 @@ github.com/chengxilo/virtualterm v1.0.4/go.mod h1:DyxxBZz/x1iqJjFxTFcr6/x+jSpqN0
|
||||
github.com/cpuguy83/go-md2man/v2 v2.0.6/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6NIQQ7OS05n1F4g=
|
||||
github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c=
|
||||
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY=
|
||||
github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto=
|
||||
github.com/google/pprof v0.0.0-20250317173921-a4b03ec1a45e h1:ijClszYn+mADRFY17kjQEVQ1XRhq2/JR1M3sGqeJoxs=
|
||||
github.com/google/pprof v0.0.0-20250317173921-a4b03ec1a45e/go.mod h1:boTsfXsheKC2y+lKOCMpSfarhxDeIzfZG1jqGcPl3cA=
|
||||
github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0=
|
||||
github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo=
|
||||
github.com/hashicorp/golang-lru/v2 v2.0.7 h1:a+bsQ5rvGLjzHuww6tVxozPZFVghXaHOwFs4luLUK2k=
|
||||
github.com/hashicorp/golang-lru/v2 v2.0.7/go.mod h1:QeFd9opnmA6QUJc5vARoKUSoFhyfM2/ZepoAG6RGpeM=
|
||||
github.com/inconshreveable/mousetrap v1.1.0 h1:wN+x4NVGpMsO7ErUn/mUI3vEoE6Jt13X2s0bqwp9tc8=
|
||||
github.com/inconshreveable/mousetrap v1.1.0/go.mod h1:vpF70FUmC8bwa3OWnCshd2FqLfsEA9PFc4w1p2J65bw=
|
||||
github.com/mattn/go-isatty v0.0.22 h1:j8l17JJ9i6VGPUFUYoTUKPSgKe/83EYU2zBC7YNKMw4=
|
||||
github.com/mattn/go-isatty v0.0.22/go.mod h1:ZXfXG4SQHsB/w3ZeOYbR0PrPwLy+n6xiMrJlRFqopa4=
|
||||
github.com/mattn/go-runewidth v0.0.16 h1:E5ScNMtiwvlvB5paMFdw9p4kSQzbXFikJ5SQO6TULQc=
|
||||
github.com/mattn/go-runewidth v0.0.16/go.mod h1:Jdepj2loyihRzMpdS35Xk/zdY8IAYHsh153qUoGf23w=
|
||||
github.com/mitchellh/colorstring v0.0.0-20190213212951-d06e56a500db h1:62I3jR2EmQ4l5rM/4FEfDWcRD+abF5XlKShorW5LRoQ=
|
||||
github.com/mitchellh/colorstring v0.0.0-20190213212951-d06e56a500db/go.mod h1:l0dey0ia/Uv7NcFFVbCLtqEBQbrT4OCwCSKTEv6enCw=
|
||||
github.com/ncruces/go-strftime v1.0.0 h1:HMFp8mLCTPp341M/ZnA4qaf7ZlsbTc+miZjCLOFAw7w=
|
||||
github.com/ncruces/go-strftime v1.0.0/go.mod h1:Fwc5htZGVVkseilnfgOVb9mKy6w1naJmn9CehxcKcls=
|
||||
github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM=
|
||||
github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94icq4NjY3clb7Lk8O1qJ8BdBEF8z0ibU0rE=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo=
|
||||
github.com/rivo/uniseg v0.4.7 h1:WUdvkW8uEhrYfLC4ZzdpI2ztxP1I582+49Oc5Mq64VQ=
|
||||
github.com/rivo/uniseg v0.4.7/go.mod h1:FN3SvrM+Zdj16jyLfmOkMNblXMcoc8DfTHruCPUcx88=
|
||||
github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM=
|
||||
@@ -23,10 +37,44 @@ github.com/spf13/pflag v1.0.9/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An
|
||||
github.com/stretchr/testify v1.9.0 h1:HtqpIVDClZ4nwg75+f6Lvsy/wHu+3BoSGCbBAcpTsTg=
|
||||
github.com/stretchr/testify v1.9.0/go.mod h1:r2ic/lqez/lEtzL7wO/rwa5dbSLXVDPFyf8C91i36aY=
|
||||
go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg=
|
||||
golang.org/x/mod v0.37.0 h1:vF1DjpVEshcIqoEaauuHebaLk1O1forxjxBaVn884JQ=
|
||||
golang.org/x/mod v0.37.0/go.mod h1:m8S8VeM9r4dzDwjrKO0a1sZP3YjeMamRRlD+fmR2Q/0=
|
||||
golang.org/x/sync v0.21.0 h1:HLII4xRRTtCRkxYp4HNFF0Js/Og6q2i++KXbg0gHCwM=
|
||||
golang.org/x/sync v0.21.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
|
||||
golang.org/x/sys v0.46.0 h1:noSf2Fq6F8DBgS+LysIkx7rIExoNHJsxOAtPp4rthXw=
|
||||
golang.org/x/sys v0.46.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/term v0.44.0 h1:0rLvDRCtNj0gZkyIXhCyOb2OAzEhLVqc4B+hrsBhrmc=
|
||||
golang.org/x/term v0.44.0/go.mod h1:7ze4MdzUzLXpSAoFP1H0bOI9aXDqveSvatT5vKcFh2Y=
|
||||
golang.org/x/tools v0.47.0 h1:7Kn5x/d1svx/PzryTsqeoZN4TZwqeH5pGWjefhLi/1Q=
|
||||
golang.org/x/tools v0.47.0/go.mod h1:dFHnyTvFWY212G+h7ZY4Vsp/K3U4/7W9TyVaAul8uCA=
|
||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
|
||||
gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
|
||||
modernc.org/cc/v4 v4.29.0 h1:CXgwL8cvxmyzBQZzbSl/6xFtMCryb6u8IOqDci39cgc=
|
||||
modernc.org/cc/v4 v4.29.0/go.mod h1:OnovgIhbbMXMu1aISnJ0wvVD1KnW+cAUJkIrAWh+kVI=
|
||||
modernc.org/ccgo/v4 v4.34.6 h1:sBgfIwyN0TQ9C5hwIeuqyeAKyMWnbvj2fvpF4L11uzU=
|
||||
modernc.org/ccgo/v4 v4.34.6/go.mod h1:SZ8YcN9NG7XVsQYdm6jYBvi8PQP1qi+kqB6OhjqI3Fk=
|
||||
modernc.org/fileutil v1.4.0 h1:j6ZzNTftVS054gi281TyLjHPp6CPHr2KCxEXjEbD6SM=
|
||||
modernc.org/fileutil v1.4.0/go.mod h1:EqdKFDxiByqxLk8ozOxObDSfcVOv/54xDs/DUHdvCUU=
|
||||
modernc.org/gc/v2 v2.6.5 h1:nyqdV8q46KvTpZlsw66kWqwXRHdjIlJOhG6kxiV/9xI=
|
||||
modernc.org/gc/v2 v2.6.5/go.mod h1:YgIahr1ypgfe7chRuJi2gD7DBQiKSLMPgBQe9oIiito=
|
||||
modernc.org/gc/v3 v3.1.4 h1:2g65LGVSmFQrXeITAw97x7hCRvZFcyE1uDP+7Vng7JI=
|
||||
modernc.org/gc/v3 v3.1.4/go.mod h1:HFK/6AGESC7Ex+EZJhJ2Gni6cTaYpSMmU/cT9RmlfYY=
|
||||
modernc.org/goabi0 v0.2.0 h1:HvEowk7LxcPd0eq6mVOAEMai46V+i7Jrj13t4AzuNks=
|
||||
modernc.org/goabi0 v0.2.0/go.mod h1:CEFRnnJhKvWT1c1JTI3Avm+tgOWbkOu5oPA8eH8LnMI=
|
||||
modernc.org/libc v1.74.1 h1:bdR4VTKFMC4966QSNZ05XLGI/VwzVa2kTUX51Dm0riQ=
|
||||
modernc.org/libc v1.74.1/go.mod h1:uH4t5bOx3G3g9Xcmj10YKlTcVISlRDwv8VoQJG9n8Os=
|
||||
modernc.org/mathutil v1.7.1 h1:GCZVGXdaN8gTqB1Mf/usp1Y/hSqgI2vAGGP4jZMCxOU=
|
||||
modernc.org/mathutil v1.7.1/go.mod h1:4p5IwJITfppl0G4sUEDtCr4DthTaT47/N3aT6MhfgJg=
|
||||
modernc.org/memory v1.11.0 h1:o4QC8aMQzmcwCK3t3Ux/ZHmwFPzE6hf2Y5LbkRs+hbI=
|
||||
modernc.org/memory v1.11.0/go.mod h1:/JP4VbVC+K5sU2wZi9bHoq2MAkCnrt2r98UGeSK7Mjw=
|
||||
modernc.org/opt v0.2.0 h1:tGyef5ApycA7FSEOMraay9SaTk5zmbx7Tu+cJs4QKZg=
|
||||
modernc.org/opt v0.2.0/go.mod h1:03fq9lsNfvkYSfxrfUhZCWPk1lm4cq4N+Bh//bEtgns=
|
||||
modernc.org/sortutil v1.2.1 h1:+xyoGf15mM3NMlPDnFqrteY07klSFxLElE2PVuWIJ7w=
|
||||
modernc.org/sortutil v1.2.1/go.mod h1:7ZI3a3REbai7gzCLcotuw9AC4VZVpYMjDzETGsSMqJE=
|
||||
modernc.org/sqlite v1.54.0 h1:JCxR4qwkJvOaqAoYcgDoO25Nc+ROg6EJ2LfBVzdrgog=
|
||||
modernc.org/sqlite v1.54.0/go.mod h1:4ntCLuNmnH8+GNqjka1wNg7KJd5/Hi5FYp8K+XQ7GZw=
|
||||
modernc.org/strutil v1.2.1 h1:UneZBkQA+DX2Rp35KcM69cSsNES9ly8mQWD71HKlOA0=
|
||||
modernc.org/strutil v1.2.1/go.mod h1:EHkiggD70koQxjVdSBM3JKM7k6L0FbGE5eymy9i3B9A=
|
||||
modernc.org/token v1.1.0 h1:Xl7Ap9dKaEs5kLoOQeQmPWevfnk/DM5qcLcYlA8ys6Y=
|
||||
modernc.org/token v1.1.0/go.mod h1:UGzOrNV1mAFSEB63lOFHIpNRUVMvYTc6yu1SMY/XTDM=
|
||||
|
||||
@@ -1,33 +1,60 @@
|
||||
// Command sfdupes quickly identifies candidate duplicate files across
|
||||
// very large filesystems without reading full file contents. Files are
|
||||
// considered duplicates when they have identical size, identical SHA-256
|
||||
// of their first 1024 bytes, and identical SHA-256 of their last 1024
|
||||
// bytes.
|
||||
// very large filesystems without reading every byte of every file.
|
||||
// Files are considered duplicates when their sizes are equal and they
|
||||
// agree on a short ladder of SHA-256 hashes. A file under 10 MiB is
|
||||
// hashed in full. A larger file is compared on the hashes of its first
|
||||
// and last 64 KiB, and only when those match another file's is its
|
||||
// content hash computed and compared: of the whole file when it is
|
||||
// under 50 MiB, or of gigabyte-spaced 1 MiB samples when it is 50 MiB
|
||||
// or larger. scan maintains a persistent SQLite database of file
|
||||
// signatures (SFDUPES_DATABASE, default /var/lib/sfdupes/db.sqlite)
|
||||
// that the reporting subcommands read.
|
||||
//
|
||||
// Usage:
|
||||
//
|
||||
// sfdupes scan [--workers N] [-x] PATH... > files.dat
|
||||
// sfdupes report [files.dat|-] > dupes.tsv
|
||||
// sfdupes trees [files.dat|-] > dupetrees.tsv
|
||||
// sfdupes scan [--workers N] [-x] PATH...
|
||||
// sfdupes report > dupes.tsv
|
||||
// sfdupes trees > dupetrees.tsv
|
||||
// sfdupes --version
|
||||
//
|
||||
// See README.md for the complete specification.
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"runtime"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
// Exit codes: 0 is success (even with per-file warnings), exitFatal is
|
||||
// a fatal error, exitUsage is a usage error.
|
||||
// Exit codes: exitOK is success (even with per-file warnings),
|
||||
// exitFatal is a fatal error, exitUsage is a usage error.
|
||||
const (
|
||||
exitOK = 0
|
||||
exitFatal = 1
|
||||
exitUsage = 2
|
||||
)
|
||||
|
||||
// The subcommand names, as typed on the command line.
|
||||
const (
|
||||
cmdScan = "scan"
|
||||
cmdReport = "report"
|
||||
cmdTrees = "trees"
|
||||
)
|
||||
|
||||
// errNoSubcommand is returned by the root command when it is invoked
|
||||
// without a subcommand. That is a usage error, and the usage text
|
||||
// cobra prints for it is the whole message.
|
||||
var errNoSubcommand = errors.New("no subcommand")
|
||||
|
||||
// errWorkersBelowOne is the usage error for a scan --workers value
|
||||
// below 1.
|
||||
var errWorkersBelowOne = errors.New("--workers must be at least 1")
|
||||
|
||||
// Version is the build version, injected at link time via -ldflags
|
||||
// (see the Makefile); "dev" for a plain go build.
|
||||
//
|
||||
@@ -35,72 +62,183 @@ const (
|
||||
var Version = "dev"
|
||||
|
||||
func main() {
|
||||
// Once the reader of a stdout pipe has gone, as in "sfdupes report |
|
||||
// head", the Go runtime ends the process with SIGPIPE on the next
|
||||
// write instead of returning an error (README "Error handling").
|
||||
// Registering for SIGPIPE with os/signal would change that.
|
||||
os.Exit(run(os.Args[1:], os.Stdout, os.Stderr))
|
||||
}
|
||||
|
||||
// run executes args against the command tree and returns the process
|
||||
// exit code. It is the program's single exit point: the subcommands
|
||||
// return their errors instead of exiting, so every deferred cleanup —
|
||||
// above all closing the database, which checkpoints the SQLite WAL —
|
||||
// runs before the process ends. The report and trees subcommands write
|
||||
// their data to stdout.
|
||||
func run(args []string, stdout, stderr io.Writer) int {
|
||||
// A nil slice makes cobra fall back to os.Args, which would let a
|
||||
// test binary's own flags reach the command tree.
|
||||
if args == nil {
|
||||
args = []string{}
|
||||
}
|
||||
|
||||
root := newRootCommand(stdout, stderr)
|
||||
root.SetArgs(args)
|
||||
|
||||
err := root.Execute()
|
||||
|
||||
var fatal fatalError
|
||||
|
||||
switch {
|
||||
case err == nil:
|
||||
return exitOK
|
||||
case errors.Is(err, errInterrupted):
|
||||
// The interrupted scan has printed its own line.
|
||||
return exitFatal
|
||||
case errors.As(err, &fatal):
|
||||
// The command ran and failed: a runtime error, reported
|
||||
// without the usage text that a usage error gets.
|
||||
_, _ = fmt.Fprintf(stderr, "sfdupes: %v\n", err)
|
||||
|
||||
return exitFatal
|
||||
default:
|
||||
// A usage error, which cobra has already reported on stderr.
|
||||
return exitUsage
|
||||
}
|
||||
}
|
||||
|
||||
// newRootCommand builds the command tree. Everything on stdout is
|
||||
// machine-readable data, the version line included; all human-facing
|
||||
// output (help, usage, errors) goes to stderr.
|
||||
func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
||||
var showVersion bool
|
||||
|
||||
printVersion := runE(func(context.Context, []string) error {
|
||||
_, err := fmt.Fprintf(stdout, "sfdupes %s\n", Version)
|
||||
if err != nil {
|
||||
return fmt.Errorf("write stdout: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
})
|
||||
|
||||
root := &cobra.Command{
|
||||
Use: "sfdupes",
|
||||
Short: "Find candidate duplicate files by size and head/tail SHA-256",
|
||||
Version: Version,
|
||||
Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
|
||||
Args: cobra.NoArgs,
|
||||
Run: func(cmd *cobra.Command, _ []string) {
|
||||
// A missing subcommand prints usage and exits 2.
|
||||
_ = cmd.Usage()
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
if showVersion {
|
||||
return printVersion(cmd, args)
|
||||
}
|
||||
|
||||
os.Exit(exitUsage)
|
||||
// A missing subcommand prints usage and exits 2: cobra
|
||||
// prints the usage text for the returned error, and run
|
||||
// maps everything that is not a fatal error to exit 2.
|
||||
cmd.SilenceErrors = true
|
||||
|
||||
return errNoSubcommand
|
||||
},
|
||||
}
|
||||
// Everything on stdout is machine-readable data; all human-facing
|
||||
// output (help, usage, errors) goes to stderr.
|
||||
root.SetOut(os.Stderr)
|
||||
root.SetErr(os.Stderr)
|
||||
root.SetOut(stderr)
|
||||
root.SetErr(stderr)
|
||||
root.CompletionOptions.DisableDefaultCmd = true
|
||||
|
||||
// Cobra's built-in version flag prints through the help writer,
|
||||
// stderr; this one prints to stdout.
|
||||
root.Flags().BoolVarP(&showVersion, "version", "v", false,
|
||||
"print the version to stdout")
|
||||
|
||||
var (
|
||||
scanWorkers int
|
||||
scanOneFS bool
|
||||
)
|
||||
|
||||
scanCmd := &cobra.Command{
|
||||
Use: "scan [--workers N] [-x] PATH...",
|
||||
Short: "Walk trees and emit one record per regular file on stdout",
|
||||
Use: cmdScan + " [--workers N] [-x] PATH...",
|
||||
Short: "Walk trees and synchronize the scan database",
|
||||
Args: cobra.MinimumNArgs(1),
|
||||
Run: func(_ *cobra.Command, args []string) {
|
||||
runScan(args, scanWorkers, scanOneFS)
|
||||
PreRunE: func(cmd *cobra.Command, _ []string) error {
|
||||
return checkScanWorkers(cmd, scanWorkers)
|
||||
},
|
||||
RunE: runE(func(ctx context.Context, args []string) error {
|
||||
ctx, stop := interruptContext(ctx)
|
||||
defer stop()
|
||||
|
||||
return runScan(ctx, args, scanWorkers, scanOneFS)
|
||||
}),
|
||||
}
|
||||
scanCmd.Flags().IntVar(&scanWorkers, "workers", runtime.NumCPU(),
|
||||
"concurrent workers for the stat and hash passes")
|
||||
"concurrent workers for the walk, hash, and content phases")
|
||||
scanCmd.Flags().BoolVarP(&scanOneFS, "one-file-system", "x", false,
|
||||
"do not cross filesystem boundaries")
|
||||
|
||||
reportCmd := &cobra.Command{
|
||||
Use: "report [files.dat|-]",
|
||||
Short: "Read a scan stream and print the file-level duplicates report",
|
||||
Args: cobra.MaximumNArgs(1),
|
||||
Run: func(_ *cobra.Command, args []string) {
|
||||
runReport(args)
|
||||
},
|
||||
Use: cmdReport,
|
||||
Short: "Read the scan database and print the file-level duplicates report",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: runE(func(ctx context.Context, _ []string) error {
|
||||
return runReport(ctx, stdout)
|
||||
}),
|
||||
}
|
||||
|
||||
treesCmd := &cobra.Command{
|
||||
Use: "trees [files.dat|-]",
|
||||
Short: "Read a scan stream and print the duplicate-tree report",
|
||||
Args: cobra.MaximumNArgs(1),
|
||||
Run: func(_ *cobra.Command, args []string) {
|
||||
runTrees(args)
|
||||
},
|
||||
Use: cmdTrees,
|
||||
Short: "Read the scan database and print the duplicate-tree report",
|
||||
Args: cobra.NoArgs,
|
||||
RunE: runE(func(ctx context.Context, _ []string) error {
|
||||
return runTrees(ctx, stdout)
|
||||
}),
|
||||
}
|
||||
|
||||
root.AddCommand(scanCmd, reportCmd, treesCmd)
|
||||
|
||||
err := root.Execute()
|
||||
return root
|
||||
}
|
||||
|
||||
// checkScanWorkers rejects a scan --workers value below 1. That is a
|
||||
// usage error reported in one line: cobra prints the returned message
|
||||
// without the usage text, and run exits 2.
|
||||
func checkScanWorkers(cmd *cobra.Command, workers int) error {
|
||||
if workers >= 1 {
|
||||
return nil
|
||||
}
|
||||
|
||||
cmd.SilenceUsage = true
|
||||
|
||||
return fmt.Errorf("%w, got %d", errWorkersBelowOne, workers)
|
||||
}
|
||||
|
||||
// runE adapts a subcommand implementation, or the version print, to
|
||||
// cobra's RunE. Cobra prints the error and the command's usage text for
|
||||
// every error RunE returns, but a subcommand that ran and failed has no
|
||||
// usage problem to report: both are silenced here, and the error is
|
||||
// marked fatal so that run reports it on stderr and exits 1 rather than
|
||||
// 2. The command's context is handed to the implementation: cancelling
|
||||
// it unwinds the scan's worker pools.
|
||||
func runE(
|
||||
fn func(ctx context.Context, args []string) error,
|
||||
) func(*cobra.Command, []string) error {
|
||||
return func(cmd *cobra.Command, args []string) error {
|
||||
cmd.SilenceUsage = true
|
||||
cmd.SilenceErrors = true
|
||||
|
||||
err := fn(cmd.Context(), args)
|
||||
if err != nil {
|
||||
// Cobra has already printed the error and usage to stderr;
|
||||
// an invalid subcommand or bad arguments is a usage error.
|
||||
os.Exit(exitUsage)
|
||||
return fatalError{err: err}
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
// fatalf reports a fatal error and exits 1.
|
||||
func fatalf(format string, args ...any) {
|
||||
fmt.Fprintf(os.Stderr, "sfdupes: "+format+"\n", args...)
|
||||
os.Exit(exitFatal)
|
||||
// fatalError marks a runtime failure, as opposed to the usage errors
|
||||
// cobra itself produces while parsing arguments and flags. Both come
|
||||
// out of Execute as plain errors, so the wrapper is what tells run to
|
||||
// report this one as "sfdupes: ..." and exit 1.
|
||||
type fatalError struct {
|
||||
err error
|
||||
}
|
||||
|
||||
func (e fatalError) Error() string { return e.err.Error() }
|
||||
|
||||
func (e fatalError) Unwrap() error { return e.err }
|
||||
|
||||
+792
@@ -0,0 +1,792 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"io"
|
||||
"io/fs"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// usageMarker is the first line of cobra's usage text, which a usage
|
||||
// error prints and a runtime failure must not.
|
||||
const usageMarker = "Usage:"
|
||||
|
||||
// walSuffixes are the SQLite sidecar files a WAL-mode database keeps
|
||||
// while it is open. A clean close checkpoints the WAL and removes
|
||||
// both; finding either afterwards means the database was never closed.
|
||||
//
|
||||
//nolint:gochecknoglobals // a constant list, immutable by convention
|
||||
var walSuffixes = []string{"-wal", "-shm"}
|
||||
|
||||
// assertNoSidecars fails when a WAL sidecar is still present beside the
|
||||
// database at path.
|
||||
func assertNoSidecars(t *testing.T, path string) {
|
||||
t.Helper()
|
||||
|
||||
for _, suffix := range walSuffixes {
|
||||
_, err := os.Stat(path + suffix)
|
||||
if err == nil {
|
||||
t.Errorf("%s%s still present: the database was not closed",
|
||||
path, suffix)
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
if !errors.Is(err, fs.ErrNotExist) {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// makeReadOnly takes write permission away from the database at path,
|
||||
// from any WAL sidecar beside it, and from their directory, as for a
|
||||
// user reading a database that a root cron scan keeps. Root ignores
|
||||
// file permissions, so it skips the test when run as root.
|
||||
func makeReadOnly(t *testing.T, path string) {
|
||||
t.Helper()
|
||||
|
||||
if os.Geteuid() == 0 {
|
||||
t.Skip("root ignores file permissions")
|
||||
}
|
||||
|
||||
err := os.Chmod(path, 0o400)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
for _, suffix := range walSuffixes {
|
||||
err = os.Chmod(path+suffix, 0o400)
|
||||
if err != nil && !errors.Is(err, fs.ErrNotExist) {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
dir := filepath.Dir(path)
|
||||
|
||||
//nolint:gosec // reaching the database needs the search bit
|
||||
err = os.Chmod(dir, 0o500)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Runs before t.TempDir's own cleanup, which must delete the files.
|
||||
t.Cleanup(func() {
|
||||
//nolint:gosec // removing the directory needs its search bit back
|
||||
_ = os.Chmod(dir, 0o700)
|
||||
})
|
||||
}
|
||||
|
||||
// captureStderr redirects os.Stderr to a file for the rest of the test
|
||||
// and returns a function reading back everything written to it. scan
|
||||
// writes its warnings and summary straight to os.Stderr, not to the
|
||||
// stderr writer run is given.
|
||||
func captureStderr(t *testing.T) func() string {
|
||||
t.Helper()
|
||||
|
||||
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
saved := os.Stderr
|
||||
os.Stderr = f
|
||||
|
||||
t.Cleanup(func() {
|
||||
os.Stderr = saved
|
||||
|
||||
_ = f.Close()
|
||||
})
|
||||
|
||||
return func() string {
|
||||
// Read what has been written without disturbing the write
|
||||
// offset, so the capture can be inspected more than once.
|
||||
size, err := f.Seek(0, io.SeekCurrent)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if size == 0 {
|
||||
return ""
|
||||
}
|
||||
|
||||
b := make([]byte, size)
|
||||
|
||||
_, err = f.ReadAt(b, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return string(b)
|
||||
}
|
||||
}
|
||||
|
||||
// brokenDatabase writes a database that opens cleanly and passes the
|
||||
// schema-version check but has no files table, so the first query
|
||||
// fails with the database already open: a fatal error on a path that
|
||||
// owns an open database. It closes the database the way scan does.
|
||||
func brokenDatabase(t *testing.T) string {
|
||||
t.Helper()
|
||||
|
||||
path := testDBPath(t)
|
||||
|
||||
db, err := openDB(path, scanParams)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
_, err = db.ExecContext(context.Background(),
|
||||
"PRAGMA user_version = "+strconv.Itoa(schemaVersion))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
closeScanDatabase(t.Context(), db, path)
|
||||
|
||||
return path
|
||||
}
|
||||
|
||||
func TestOpenDatabaseKeepsWALWhileOpen(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// The premise of the fatal-path tests below: an open database has
|
||||
// a -wal sidecar, so its absence afterwards is evidence that the
|
||||
// database was closed and its WAL checkpointed.
|
||||
path := testDBPath(t)
|
||||
|
||||
db, err := openScanDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
_, err = os.Stat(path + "-wal")
|
||||
if err != nil {
|
||||
t.Fatalf("no -wal beside an open database: %v", err)
|
||||
}
|
||||
|
||||
err = db.Close()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
assertNoSidecars(t, path)
|
||||
}
|
||||
|
||||
func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
|
||||
// Every subcommand that owns an open database must close it when
|
||||
// it fails: no os.Exit between the open and the return. The
|
||||
// sidecar check is evidence of the close only for scan: report and
|
||||
// trees only read a database that is out of WAL mode, which leaves
|
||||
// nothing on disk whether they close it or not.
|
||||
cases := map[string][]string{
|
||||
cmdScan: {cmdScan},
|
||||
cmdReport: {cmdReport},
|
||||
cmdTrees: {cmdTrees},
|
||||
}
|
||||
|
||||
for name, args := range cases {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
path := brokenDatabase(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
if name == cmdScan {
|
||||
args = append(args, t.TempDir())
|
||||
}
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run(args, &stdout, &stderr)
|
||||
if code != exitFatal {
|
||||
t.Errorf("run(%v) = %d, want %d", args, code, exitFatal)
|
||||
}
|
||||
|
||||
assertNoSidecars(t, path)
|
||||
assertFatalOutput(t, stderr.String(), stdout.String())
|
||||
|
||||
// Proof that the failure happened after the open: only a
|
||||
// query against the opened database can report this.
|
||||
if !strings.Contains(stderr.String(), "no such table: files") {
|
||||
t.Errorf("stderr = %q, want the failure to come from a "+
|
||||
"query on the open database", stderr.String())
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunMissingOperandIsFatalNotUsage(t *testing.T) {
|
||||
// README §Error handling: a PATH operand that does not exist is a
|
||||
// fatal error (1), not a usage error (2) — and a runtime failure
|
||||
// must not dump the usage text.
|
||||
t.Setenv(databaseEnv, testDBPath(t))
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
missing := filepath.Join(t.TempDir(), "nope")
|
||||
|
||||
code := run([]string{cmdScan, missing}, &stdout, &stderr)
|
||||
if code != exitFatal {
|
||||
t.Errorf("run(scan %s) = %d, want %d", missing, code, exitFatal)
|
||||
}
|
||||
|
||||
assertFatalOutput(t, stderr.String(), stdout.String())
|
||||
}
|
||||
|
||||
// assertFatalOutput checks that a fatal error was reported the way
|
||||
// README §Error handling and design goal 4 require: the message on
|
||||
// stderr, prefixed with the program name, no usage text, and nothing
|
||||
// at all on stdout.
|
||||
func assertFatalOutput(t *testing.T, stderr, stdout string) {
|
||||
t.Helper()
|
||||
|
||||
if !strings.Contains(stderr, "sfdupes: ") {
|
||||
t.Errorf("stderr = %q, want a \"sfdupes: \" error report", stderr)
|
||||
}
|
||||
|
||||
if strings.Contains(stderr, usageMarker) {
|
||||
t.Errorf("stderr = %q, want no usage text for a runtime failure",
|
||||
stderr)
|
||||
}
|
||||
|
||||
if stdout != "" {
|
||||
t.Errorf("stdout = %q, want nothing (data only)", stdout)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunUsageErrors(t *testing.T) {
|
||||
// Usage errors keep exiting 2 with cobra's own report on stderr.
|
||||
cases := map[string]struct {
|
||||
args []string
|
||||
want string
|
||||
}{
|
||||
"no subcommand": {[]string{}, usageMarker},
|
||||
"scan without paths": {[]string{cmdScan}, usageMarker},
|
||||
"report with args": {[]string{cmdReport, "x"}, usageMarker},
|
||||
"trees with args": {[]string{cmdTrees, "x"}, usageMarker},
|
||||
"unknown flag": {[]string{cmdScan, "--nope", "/"}, usageMarker},
|
||||
"unknown subcommand": {[]string{"nope"}, "unknown command"},
|
||||
}
|
||||
|
||||
for name, tc := range cases {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
// No usage error may reach the database, so point it at a
|
||||
// path that does not exist.
|
||||
t.Setenv(databaseEnv, testDBPath(t))
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run(tc.args, &stdout, &stderr)
|
||||
if code != exitUsage {
|
||||
t.Errorf("run(%v) = %d, want %d", tc.args, code, exitUsage)
|
||||
}
|
||||
|
||||
if !strings.Contains(stderr.String(), tc.want) {
|
||||
t.Errorf("stderr = %q, want %q", stderr.String(), tc.want)
|
||||
}
|
||||
|
||||
if got := stdout.String(); got != "" {
|
||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunScanRejectsWorkersBelowOne(t *testing.T) {
|
||||
// README §scan mode: --workers below 1 is a usage error reported in
|
||||
// one line on stderr, before the scan opens the database.
|
||||
for _, workers := range []string{"0", "-1"} {
|
||||
t.Run(workers, func(t *testing.T) {
|
||||
dbPath := testDBPath(t)
|
||||
t.Setenv(databaseEnv, dbPath)
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
args := []string{cmdScan, "--workers", workers, t.TempDir()}
|
||||
|
||||
code := run(args, &stdout, &stderr)
|
||||
if code != exitUsage {
|
||||
t.Errorf("run(%v) = %d, want %d", args, code, exitUsage)
|
||||
}
|
||||
|
||||
want := "Error: --workers must be at least 1, got " + workers +
|
||||
"\n"
|
||||
if got := stderr.String(); got != want {
|
||||
t.Errorf("stderr = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
if got := stdout.String(); got != "" {
|
||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
||||
}
|
||||
|
||||
_, err := os.Stat(dbPath)
|
||||
if !errors.Is(err, fs.ErrNotExist) {
|
||||
t.Errorf("stat %s: %v, want the database never created",
|
||||
dbPath, err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunHelp(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// README §Subcommands: help goes to stderr, exits 0, and leaves
|
||||
// stdout empty.
|
||||
cases := [][]string{{"--help"}, {"-h"}, {cmdScan, "--help"}}
|
||||
|
||||
for _, args := range cases {
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run(args, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Errorf("run(%v) = %d, want %d", args, code, exitOK)
|
||||
}
|
||||
|
||||
if !strings.Contains(stderr.String(), usageMarker) {
|
||||
t.Errorf("run(%v) stderr = %q, want the help text",
|
||||
args, stderr.String())
|
||||
}
|
||||
|
||||
if got := stdout.String(); got != "" {
|
||||
t.Errorf("run(%v) stdout = %q, want nothing (data only)",
|
||||
args, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunVersion(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// README §Subcommands: the version is one line on stdout, with
|
||||
// nothing on stderr, and exits 0.
|
||||
for _, arg := range []string{"--version", "-v"} {
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{arg}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Errorf("run(%s) = %d, want %d", arg, code, exitOK)
|
||||
}
|
||||
|
||||
want := "sfdupes " + Version + "\n"
|
||||
if got := stdout.String(); got != want {
|
||||
t.Errorf("run(%s) stdout = %q, want %q", arg, got, want)
|
||||
}
|
||||
|
||||
if got := stderr.String(); got != "" {
|
||||
t.Errorf("run(%s) stderr = %q, want nothing", arg, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunVersionWriteFailureIsFatal(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// README §Error handling: a stdout write failure exits 1, reported
|
||||
// in one line on stderr.
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run([]string{"--version"}, failingWriter{}, &stderr)
|
||||
if code != exitFatal {
|
||||
t.Errorf("run(--version) = %d, want %d", code, exitFatal)
|
||||
}
|
||||
|
||||
want := "sfdupes: write stdout: " + errWriteFailed.Error() + "\n"
|
||||
if got := stderr.String(); got != want {
|
||||
t.Errorf("stderr = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// scanFixture builds a small tree holding one duplicate pair and one
|
||||
// unreadable file and scans it into the database the caller has
|
||||
// pointed SFDUPES_DATABASE at. The unreadable file makes the scan warn
|
||||
// and skip, which README §Error handling still calls a successful run.
|
||||
// It returns the duplicate pair's paths.
|
||||
func scanFixture(t *testing.T) []string {
|
||||
t.Helper()
|
||||
|
||||
dir := t.TempDir()
|
||||
|
||||
dupes := []string{
|
||||
writeFile(t, dir, "one/a.bin", pattern(1, 300)),
|
||||
writeFile(t, dir, "two/a.bin", pattern(1, 300)),
|
||||
}
|
||||
|
||||
// Same size as the pair, so the scan queues it for hashing and the
|
||||
// read fails.
|
||||
unreadable := writeFile(t, dir, "unreadable.bin", pattern(2, 300))
|
||||
|
||||
err := os.Chmod(unreadable, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
scanOK(t, dir)
|
||||
|
||||
return dupes
|
||||
}
|
||||
|
||||
// scanOK runs scan over operands, fails the test unless it exits 0 with
|
||||
// nothing on stdout, and returns everything it printed to stderr.
|
||||
func scanOK(t *testing.T, operands ...string) string {
|
||||
t.Helper()
|
||||
|
||||
var stdout bytes.Buffer
|
||||
|
||||
stderr := captureStderr(t)
|
||||
|
||||
code := run(append([]string{cmdScan}, operands...), &stdout, os.Stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(scan %q) = %d, want %d; stderr: %s",
|
||||
operands, code, exitOK, stderr())
|
||||
}
|
||||
|
||||
if got := stdout.String(); got != "" {
|
||||
t.Errorf("scan stdout = %q, want nothing (data only)", got)
|
||||
}
|
||||
|
||||
return stderr()
|
||||
}
|
||||
|
||||
func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
scanFixture(t)
|
||||
assertNoSidecars(t, path)
|
||||
}
|
||||
|
||||
func TestRunScanSkipsSymlinkOperand(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dir := t.TempDir()
|
||||
writeFile(t, dir, "target/sub/f", pattern(1, 10))
|
||||
|
||||
link := filepath.Join(dir, "link")
|
||||
|
||||
err := os.Symlink(filepath.Join(dir, "target"), link)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Scanning a directory through the symlink stores a record beneath
|
||||
// the symlink's own path for a file beneath its target.
|
||||
scanOK(t, filepath.Join(link, "sub"))
|
||||
|
||||
assertOperandSkipped(t, path, link, "symlink",
|
||||
filepath.Join(link, "sub", "f"))
|
||||
}
|
||||
|
||||
func TestRunScanWalksOperandUnderSymlinkOperand(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dir := t.TempDir()
|
||||
writeFile(t, dir, "target/sub/f", pattern(1, 10))
|
||||
|
||||
link := filepath.Join(dir, "link")
|
||||
|
||||
err := os.Symlink(filepath.Join(dir, "target"), link)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// link is dropped as a symlink, but link/sub must still be scanned,
|
||||
// not dropped as lying under link.
|
||||
scanOK(t, link, filepath.Join(link, "sub"))
|
||||
|
||||
db, err := openDB(path, reportParams)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
t.Cleanup(func() { _ = db.Close() })
|
||||
|
||||
recordByPath(t, dbRecords(t, db), filepath.Join(link, "sub", "f"))
|
||||
}
|
||||
|
||||
func TestRunScanSkipsZFSOperand(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
zfs := filepath.Join(t.TempDir(), ".zfs")
|
||||
snapshot := filepath.Join(zfs, "snapshot", "hourly")
|
||||
f := writeFile(t, snapshot, "f", pattern(1, 10))
|
||||
|
||||
// An operand beneath a .zfs directory is walked, because it is not
|
||||
// itself named .zfs.
|
||||
scanOK(t, snapshot)
|
||||
|
||||
assertOperandSkipped(t, path, zfs, ".zfs directory", f)
|
||||
}
|
||||
|
||||
// assertOperandSkipped scans operand alone and checks that it is skipped
|
||||
// as kind: a warning naming it, one skip in the summary, exit 0, and the
|
||||
// record for kept, which an earlier scan stored beneath operand, still
|
||||
// in the database at dbPath.
|
||||
func assertOperandSkipped(t *testing.T, dbPath, operand, kind,
|
||||
kept string,
|
||||
) {
|
||||
t.Helper()
|
||||
|
||||
stderr := scanOK(t, operand)
|
||||
|
||||
warning := "walk " + operand + ": skipping " + kind + " operand\n"
|
||||
if !strings.Contains(stderr, warning) {
|
||||
t.Errorf("stderr = %q, want %q", stderr, warning)
|
||||
}
|
||||
|
||||
summary := "scan: 0 files seen (0 added, 0 updated, 0 removed, " +
|
||||
"0 unchanged), 1 skipped\n"
|
||||
if !strings.Contains(stderr, summary) {
|
||||
t.Errorf("stderr = %q, want %q", stderr, summary)
|
||||
}
|
||||
|
||||
db, err := openDB(dbPath, reportParams)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
t.Cleanup(func() { _ = db.Close() })
|
||||
|
||||
recordByPath(t, dbRecords(t, db), kept)
|
||||
}
|
||||
|
||||
func TestRunReportSucceeds(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dupes := scanFixture(t)
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{cmdReport}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
}
|
||||
|
||||
want := "first\tdupe\tsize\n" + dupes[0] + "\t" + dupes[1] + "\t300\n"
|
||||
if got := stdout.String(); got != want {
|
||||
t.Errorf("stdout = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
assertNoSidecars(t, path)
|
||||
}
|
||||
|
||||
func TestRunTreesSucceeds(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dupes := scanFixture(t)
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{cmdTrees}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
}
|
||||
|
||||
// The two directories holding the duplicate pair are duplicate
|
||||
// trees of each other.
|
||||
want := "first\tdupe\tfiles\tsize\n" +
|
||||
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) + "\t1\t300\n"
|
||||
if got := stdout.String(); got != want {
|
||||
t.Errorf("stdout = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
assertNoSidecars(t, path)
|
||||
}
|
||||
|
||||
func TestRunReportsNeedOnlyReadAccess(t *testing.T) {
|
||||
// README §Database: report and trees need only read access to the
|
||||
// database file. With its directory read-only as well, SQLite
|
||||
// cannot create any file beside it.
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
dupes := scanFixture(t)
|
||||
assertNoSidecars(t, path)
|
||||
makeReadOnly(t, path)
|
||||
|
||||
cases := map[string]string{
|
||||
cmdReport: "first\tdupe\tsize\n" +
|
||||
dupes[0] + "\t" + dupes[1] + "\t300\n",
|
||||
cmdTrees: "first\tdupe\tfiles\tsize\n" +
|
||||
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) +
|
||||
"\t1\t300\n",
|
||||
}
|
||||
|
||||
for name, want := range cases {
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{name}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Errorf("run(%s) = %d, want %d; stderr: %s",
|
||||
name, code, exitOK, stderr.String())
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
if got := stdout.String(); got != want {
|
||||
t.Errorf("%s stdout = %q, want %q", name, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// holdScanLock takes the lock on the database at path, as a running
|
||||
// scan does, and holds it until the test ends. It fails the test when
|
||||
// the lock is already held.
|
||||
func holdScanLock(t *testing.T, path string) {
|
||||
t.Helper()
|
||||
|
||||
lock, err := lockScanDatabase(path)
|
||||
if err != nil {
|
||||
t.Fatalf("lock %s: %v", path, err)
|
||||
}
|
||||
|
||||
t.Cleanup(func() { _ = lock.Close() })
|
||||
}
|
||||
|
||||
func TestRunSecondScanFails(t *testing.T) {
|
||||
// README §Database: while one scan holds the lock, a second scan
|
||||
// fails at once, naming the lock file, without creating the
|
||||
// database.
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
holdScanLock(t, path)
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{cmdScan, t.TempDir()}, &stdout, &stderr)
|
||||
if code != exitFatal {
|
||||
t.Errorf("run(scan) = %d, want %d", code, exitFatal)
|
||||
}
|
||||
|
||||
want := "sfdupes: another scan is running (lock held on " +
|
||||
path + ".lock)\n"
|
||||
if got := stderr.String(); got != want {
|
||||
t.Errorf("stderr = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
if got := stdout.String(); got != "" {
|
||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
||||
}
|
||||
|
||||
_, err := os.Stat(path)
|
||||
if !errors.Is(err, fs.ErrNotExist) {
|
||||
t.Errorf("stat %s = %v, want the database not created", path, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunScanReleasesLock(t *testing.T) {
|
||||
// README §Database: a scan releases the lock however it ends.
|
||||
t.Run("success", func(t *testing.T) {
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
scanFixture(t)
|
||||
holdScanLock(t, path)
|
||||
})
|
||||
|
||||
t.Run("fatal error", func(t *testing.T) {
|
||||
path := brokenDatabase(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
code := run([]string{cmdScan, t.TempDir()}, io.Discard, io.Discard)
|
||||
if code != exitFatal {
|
||||
t.Fatalf("run(scan) = %d, want %d", code, exitFatal)
|
||||
}
|
||||
|
||||
holdScanLock(t, path)
|
||||
})
|
||||
}
|
||||
|
||||
func TestRunReportsDuringScan(t *testing.T) {
|
||||
// README §Database: report and trees never take the lock, so they
|
||||
// run while a scan holds it.
|
||||
path := testDBPath(t)
|
||||
t.Setenv(databaseEnv, path)
|
||||
|
||||
scanFixture(t)
|
||||
holdScanLock(t, path)
|
||||
|
||||
for _, name := range []string{cmdReport, cmdTrees} {
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run([]string{name}, io.Discard, &stderr)
|
||||
if code != exitOK {
|
||||
t.Errorf("run(%s) = %d, want %d; stderr: %s",
|
||||
name, code, exitOK, stderr.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunStdoutClosedIsFatal(t *testing.T) {
|
||||
// README §Error handling: a stdout write failure exits 1, reported
|
||||
// in one line on stderr.
|
||||
for _, name := range []string{cmdReport, cmdTrees} {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
t.Setenv(databaseEnv, testDBPath(t))
|
||||
|
||||
scanFixture(t)
|
||||
|
||||
stdout, err := os.Create(filepath.Join(t.TempDir(), "stdout"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = stdout.Close()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
var stderr bytes.Buffer
|
||||
|
||||
code := run([]string{name}, stdout, &stderr)
|
||||
if code != exitFatal {
|
||||
t.Errorf("run(%s) = %d, want %d", name, code, exitFatal)
|
||||
}
|
||||
|
||||
got := stderr.String()
|
||||
if !strings.HasPrefix(got, "sfdupes: write stdout: ") ||
|
||||
!strings.Contains(got, os.ErrClosed.Error()) ||
|
||||
strings.Count(got, "\n") != 1 {
|
||||
t.Errorf("stderr = %q, want one line reporting the "+
|
||||
"failed stdout write", got)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// errWriteFailed is the error failingWriter returns.
|
||||
var errWriteFailed = errors.New("write failed")
|
||||
|
||||
// failingWriter is a stdout that fails every write.
|
||||
type failingWriter struct{}
|
||||
|
||||
func (failingWriter) Write([]byte) (int, error) { return 0, errWriteFailed }
|
||||
|
||||
func TestStdoutWriteErrorPropagates(t *testing.T) {
|
||||
t.Setenv(databaseEnv, testDBPath(t))
|
||||
|
||||
scanFixture(t)
|
||||
|
||||
cases := map[string]func(context.Context, io.Writer) error{
|
||||
cmdReport: runReport,
|
||||
cmdTrees: runTrees,
|
||||
}
|
||||
|
||||
for name, fn := range cases {
|
||||
err := fn(t.Context(), failingWriter{})
|
||||
if !errors.Is(err, errWriteFailed) {
|
||||
t.Errorf("%s: error = %v, want %v", name, err, errWriteFailed)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"devDependencies": {
|
||||
"prettier": "3.8.1"
|
||||
}
|
||||
}
|
||||
+61
-14
@@ -6,6 +6,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/schollz/progressbar/v3"
|
||||
"golang.org/x/term"
|
||||
)
|
||||
|
||||
// plainInterval is the minimum time between progress lines when stderr
|
||||
@@ -24,21 +25,23 @@ const percentScale = 100
|
||||
|
||||
// stderrIsTTY reports whether stderr is attached to a terminal.
|
||||
func stderrIsTTY() bool {
|
||||
fi, err := os.Stderr.Stat()
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
|
||||
return fi.Mode()&os.ModeCharDevice != 0
|
||||
return term.IsTerminal(int(os.Stderr.Fd()))
|
||||
}
|
||||
|
||||
// progress renders one scan pass's progress on stderr. On a TTY it
|
||||
// delegates to the progressbar library (spinner style when the total is
|
||||
// unknown, full bar with count/percent/rate/elapsed/ETA otherwise). When
|
||||
// stderr is not a TTY it emits no ANSI redraws: it prints a plain
|
||||
// one-line update no more often than every plainInterval.
|
||||
// one-line update as the pass starts, then no more often than every
|
||||
// plainInterval.
|
||||
//
|
||||
// All methods must be called from the main goroutine only.
|
||||
// All methods must be called from the main goroutine only. On a TTY
|
||||
// the library also redraws a spinner from its own goroutine, several
|
||||
// times a second, so its count and elapsed time stay current while a
|
||||
// pass waits for its next item. A nil *progress is a valid
|
||||
// no-display receiver: every method is a no-op, so batched database
|
||||
// flushes during the streaming pass can reuse the update-pass helpers
|
||||
// without rendering anything.
|
||||
type progress struct {
|
||||
label string
|
||||
total int64 // -1 when unknown (walk pass)
|
||||
@@ -50,10 +53,22 @@ type progress struct {
|
||||
|
||||
func newProgress(label string, total int64) *progress {
|
||||
p := &progress{label: label, total: total, start: time.Now()}
|
||||
if !stderrIsTTY() {
|
||||
if stderrIsTTY() {
|
||||
p.bar = newBar(label, total)
|
||||
|
||||
return p
|
||||
}
|
||||
|
||||
// Print the zero state at once: the first item may take minutes,
|
||||
// and a pass must never look hung.
|
||||
p.last = p.start
|
||||
fmt.Fprintln(os.Stderr, p.plainLine())
|
||||
|
||||
return p
|
||||
}
|
||||
|
||||
// newBar builds the TTY display for newProgress.
|
||||
func newBar(label string, total int64) *progressbar.ProgressBar {
|
||||
opts := []progressbar.Option{
|
||||
progressbar.OptionSetWriter(os.Stderr),
|
||||
progressbar.OptionSetDescription(label),
|
||||
@@ -62,6 +77,9 @@ func newProgress(label string, total int64) *progress {
|
||||
progressbar.OptionSetItsString("files"),
|
||||
progressbar.OptionSetElapsedTime(true),
|
||||
progressbar.OptionThrottle(barThrottle),
|
||||
// Render at zero immediately: a phase must be visible the
|
||||
// moment it starts, even before its first item completes.
|
||||
progressbar.OptionSetRenderBlankState(true),
|
||||
}
|
||||
if total >= 0 {
|
||||
opts = append(opts,
|
||||
@@ -75,13 +93,15 @@ func newProgress(label string, total int64) *progress {
|
||||
)
|
||||
}
|
||||
|
||||
p.bar = progressbar.NewOptions64(total, opts...)
|
||||
|
||||
return p
|
||||
return progressbar.NewOptions64(total, opts...)
|
||||
}
|
||||
|
||||
// increment records one completed item and refreshes the display.
|
||||
func (p *progress) increment() {
|
||||
if p == nil {
|
||||
return
|
||||
}
|
||||
|
||||
p.count++
|
||||
if p.bar != nil {
|
||||
_ = p.bar.Add(1)
|
||||
@@ -96,18 +116,45 @@ func (p *progress) increment() {
|
||||
}
|
||||
|
||||
// warnf prints a one-line warning to stderr without corrupting the bar.
|
||||
// The whole message is escaped like a report's path columns, so a path
|
||||
// holding a newline cannot split the warning.
|
||||
func (p *progress) warnf(format string, args ...any) {
|
||||
if p == nil {
|
||||
return
|
||||
}
|
||||
|
||||
msg := escapePath(fmt.Sprintf(format, args...))
|
||||
|
||||
if p.bar != nil && p.total < 0 {
|
||||
// The library also redraws a spinner from its own goroutine, so
|
||||
// a direct write could land inside a redraw. The bar prints the
|
||||
// warning itself, just before its next redraw.
|
||||
_, _ = progressbar.Bprintln(p.bar, msg)
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
if p.bar != nil {
|
||||
_ = p.bar.Clear()
|
||||
}
|
||||
|
||||
fmt.Fprintf(os.Stderr, format+"\n", args...)
|
||||
fmt.Fprintln(os.Stderr, msg)
|
||||
}
|
||||
|
||||
// finish terminates the pass's display.
|
||||
// finish terminates the pass's display. A bar whose pass stopped short
|
||||
// of its total, as an interrupted one does, is left as last drawn; the
|
||||
// library's Finish would fill it up.
|
||||
func (p *progress) finish() {
|
||||
if p == nil {
|
||||
return
|
||||
}
|
||||
|
||||
if p.bar != nil {
|
||||
if p.total >= 0 && p.count < p.total {
|
||||
_ = p.bar.Exit()
|
||||
} else {
|
||||
_ = p.bar.Finish()
|
||||
}
|
||||
|
||||
fmt.Fprintln(os.Stderr)
|
||||
|
||||
|
||||
@@ -0,0 +1,189 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// spinnerIdle comfortably outlasts the 100ms interval at which the
|
||||
// progressbar library redraws a spinner from its own goroutine.
|
||||
const spinnerIdle = 500 * time.Millisecond
|
||||
|
||||
//nolint:paralleltest // replaces the process-wide os.Stderr
|
||||
func TestStderrIsTTYFalseForNonTerminals(t *testing.T) {
|
||||
r, pipe, err := os.Pipe()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
regular, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
devNull, err := os.OpenFile(os.DevNull, os.O_WRONLY, 0)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
saved := os.Stderr
|
||||
|
||||
t.Cleanup(func() {
|
||||
os.Stderr = saved
|
||||
|
||||
for _, f := range []*os.File{r, pipe, regular, devNull} {
|
||||
_ = f.Close()
|
||||
}
|
||||
})
|
||||
|
||||
cases := map[string]*os.File{
|
||||
"a pipe": pipe,
|
||||
"a regular file": regular,
|
||||
os.DevNull: devNull,
|
||||
}
|
||||
|
||||
for name, f := range cases {
|
||||
os.Stderr = f
|
||||
|
||||
if stderrIsTTY() {
|
||||
t.Errorf("stderrIsTTY() = true with stderr on %s", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestNewProgressPrintsBeforeFirstItem checks that each pass shows its
|
||||
// zero state the moment it starts when stderr is not a terminal, and
|
||||
// that the next line still waits for plainInterval.
|
||||
//
|
||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
||||
func TestNewProgressPrintsBeforeFirstItem(t *testing.T) {
|
||||
stderr := captureStderr(t)
|
||||
|
||||
newProgress("walk", -1).increment()
|
||||
newProgress("hash", 10).increment()
|
||||
|
||||
want := "walk: 0 files, elapsed 0s\n" +
|
||||
"hash: [0/10] 0% 0 files/s elapsed 0s eta ?\n"
|
||||
if got := stderr(); got != want {
|
||||
t.Errorf("stderr = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// newWalkSpinner returns the walk pass's terminal display, writing to
|
||||
// os.Stderr whether or not it is a terminal, and stops the library's
|
||||
// redraws when the test ends.
|
||||
func newWalkSpinner(t *testing.T) *progress {
|
||||
t.Helper()
|
||||
|
||||
p := &progress{
|
||||
label: "walk", total: -1, start: time.Now(),
|
||||
bar: newBar("walk", -1),
|
||||
}
|
||||
t.Cleanup(p.finish)
|
||||
|
||||
return p
|
||||
}
|
||||
|
||||
// TestProgressWarningsOnOwnLines drives the terminal display of the walk
|
||||
// pass through a run of warnings with no items between them, as when the
|
||||
// walk meets many unreadable paths, for several of the spinner's
|
||||
// redraws: every warning must land on a line of its own, never inside a
|
||||
// redraw.
|
||||
//
|
||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
||||
func TestProgressWarningsOnOwnLines(t *testing.T) {
|
||||
stderr := captureStderr(t)
|
||||
p := newWalkSpinner(t)
|
||||
|
||||
// No pause between warnings: one written straight to stderr is
|
||||
// garbled only if a redraw lands while it is being written.
|
||||
issued := 0
|
||||
for start := time.Now(); time.Since(start) < spinnerIdle; issued++ {
|
||||
p.warnf("warning")
|
||||
}
|
||||
|
||||
// The spinner prints the warnings at its next redraw.
|
||||
time.Sleep(spinnerIdle)
|
||||
|
||||
// A terminal shows each line as the text after its last carriage
|
||||
// return.
|
||||
shown := 0
|
||||
|
||||
for line := range strings.SplitSeq(stderr(), "\n") {
|
||||
if !strings.Contains(line, "warning") {
|
||||
continue
|
||||
}
|
||||
|
||||
shown++
|
||||
|
||||
if text := line[strings.LastIndex(line, "\r")+1:]; text != "warning" {
|
||||
t.Errorf("terminal shows %q, want %q", text, "warning")
|
||||
}
|
||||
}
|
||||
|
||||
if shown != issued {
|
||||
t.Errorf("%d warning lines, want %d", shown, issued)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSpinnerShowsCountAfterBurst checks that once a burst of items
|
||||
// faster than the redraw limit is over, the walk display shows every
|
||||
// item completed while it waits for the next one.
|
||||
//
|
||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
||||
func TestSpinnerShowsCountAfterBurst(t *testing.T) {
|
||||
stderr := captureStderr(t)
|
||||
p := newWalkSpinner(t)
|
||||
|
||||
for range 50 {
|
||||
p.increment()
|
||||
}
|
||||
|
||||
time.Sleep(spinnerIdle)
|
||||
|
||||
if shown := lastFrame(stderr()); !strings.Contains(shown, "(50/-,") {
|
||||
t.Errorf("terminal shows %q, want a count of 50", shown)
|
||||
}
|
||||
}
|
||||
|
||||
// lastFrame returns what a terminal shows of the frames a bar drew: the
|
||||
// last one. The library starts each frame with a carriage return and
|
||||
// erases the previous one with spaces first.
|
||||
func lastFrame(out string) string {
|
||||
var shown string
|
||||
|
||||
for frame := range strings.SplitSeq(out, "\r") {
|
||||
if strings.TrimSpace(frame) != "" {
|
||||
shown = frame
|
||||
}
|
||||
}
|
||||
|
||||
return shown
|
||||
}
|
||||
|
||||
// TestBarStoppedShortKeepsCount checks that the terminal display of a
|
||||
// pass that stops before its total, as an interrupted one does, is left
|
||||
// as last drawn instead of being filled up.
|
||||
//
|
||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
||||
func TestBarStoppedShortKeepsCount(t *testing.T) {
|
||||
stderr := captureStderr(t)
|
||||
p := &progress{
|
||||
label: "hash", total: 10, start: time.Now(),
|
||||
bar: newBar("hash", 10),
|
||||
}
|
||||
|
||||
p.increment()
|
||||
|
||||
// Past the redraw limit, so the bar draws the next count.
|
||||
time.Sleep(2 * barThrottle)
|
||||
p.increment()
|
||||
p.finish()
|
||||
|
||||
if shown := lastFrame(stderr()); !strings.Contains(shown, "(2/10,") {
|
||||
t.Errorf("terminal shows %q, want a count of 2 of 10", shown)
|
||||
}
|
||||
}
|
||||
@@ -2,225 +2,113 @@ package main
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"slices"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// ioBufSize is the buffer size for the buffered scan-stream reader and
|
||||
// the buffered stdout writers.
|
||||
// ioBufSize is the buffer size for the buffered stdout writers.
|
||||
const ioBufSize = 1 << 20
|
||||
|
||||
// recordFieldCount is the number of tab-separated fields in a scan
|
||||
// record: size, mtime, head hash, tail hash, path.
|
||||
const recordFieldCount = 5
|
||||
|
||||
// scanInitBufSize and scanMaxRecordSize bound the scanner buffer used
|
||||
// to read scan records; a record longer than scanMaxRecordSize is a
|
||||
// fatal read error.
|
||||
const (
|
||||
scanInitBufSize = 64 << 10
|
||||
scanMaxRecordSize = 4 << 20
|
||||
)
|
||||
|
||||
// minGroupSize is the smallest number of members that makes a
|
||||
// duplicate group.
|
||||
const minGroupSize = 2
|
||||
|
||||
// scanRec is one well-formed record parsed from a scan stream. The
|
||||
// signature (size, head, tail) is the duplicate key; mtime is
|
||||
// informational and not retained.
|
||||
// scanRec is one file record from the database. The signature (size,
|
||||
// head, tail, content) is the duplicate key; mtime is informational
|
||||
// only and used by scan for change detection.
|
||||
type scanRec struct {
|
||||
size int64
|
||||
mtime int64
|
||||
head string
|
||||
tail string
|
||||
content string
|
||||
path string
|
||||
}
|
||||
|
||||
// openScanInput resolves the analysis-mode input: the file named by the
|
||||
// single optional positional argument, or stdin when it is absent or
|
||||
// "-". The returned closer must be called when reading is done.
|
||||
func openScanInput(args []string) (io.Reader, string, func()) {
|
||||
if len(args) == 1 && args[0] != "-" {
|
||||
f, err := os.Open(args[0])
|
||||
// runReport implements the report subcommand: it prints the file-level
|
||||
// duplicates report as TSV on stdout. SQLite groups and orders the
|
||||
// records, and each row is written as it is read, so no group is held
|
||||
// in memory. It never touches the scanned filesystem; its only I/O is
|
||||
// the database (with SQLite's temporary sort file), stdout, and stderr.
|
||||
// Any database problem, including a missing database, is fatal.
|
||||
func runReport(ctx context.Context, stdout io.Writer) error {
|
||||
dbPath := databasePath()
|
||||
|
||||
db, err := openReportDatabase(ctx, dbPath)
|
||||
if err != nil {
|
||||
fatalf("open %s: %v", args[0], err)
|
||||
return err
|
||||
}
|
||||
|
||||
return f, args[0], func() { _ = f.Close() }
|
||||
}
|
||||
defer func() { _ = db.Close() }()
|
||||
|
||||
return os.Stdin, "stdin", func() {}
|
||||
}
|
||||
out := bufio.NewWriterSize(stdout, ioBufSize)
|
||||
|
||||
// parseScanStream reads every NUL-terminated record from in. A record
|
||||
// that does not have exactly 5 fields or whose size is non-numeric is
|
||||
// counted as malformed and skipped. Total records read is
|
||||
// len(recs) + malformed.
|
||||
func parseScanStream(in io.Reader, name string) ([]scanRec, int) {
|
||||
sc := bufio.NewScanner(bufio.NewReaderSize(in, ioBufSize))
|
||||
sc.Buffer(make([]byte, 0, scanInitBufSize), scanMaxRecordSize)
|
||||
sc.Split(splitNUL)
|
||||
_, err = fmt.Fprintln(out, "first\tdupe\tsize")
|
||||
if err != nil {
|
||||
return fmt.Errorf("write stdout: %w", err)
|
||||
}
|
||||
|
||||
var (
|
||||
recs []scanRec
|
||||
malformed int
|
||||
groups, dupeFiles int
|
||||
reclaimable int64
|
||||
writeErr error
|
||||
)
|
||||
|
||||
for sc.Scan() {
|
||||
// The path is the last field and may itself contain tabs,
|
||||
// so split into at most recordFieldCount fields.
|
||||
fields := strings.SplitN(sc.Text(), "\t", recordFieldCount)
|
||||
if len(fields) != recordFieldCount {
|
||||
malformed++
|
||||
records, err := loadDupeRows(ctx, db,
|
||||
func(first, path string, size int64) error {
|
||||
// A group's first path is its first row; every other
|
||||
// path is a dupe.
|
||||
if path == first {
|
||||
groups++
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
size, err := strconv.ParseInt(fields[0], 10, 64)
|
||||
if err != nil {
|
||||
malformed++
|
||||
|
||||
continue
|
||||
}
|
||||
|
||||
recs = append(recs, scanRec{
|
||||
size: size,
|
||||
head: fields[2],
|
||||
tail: fields[3],
|
||||
path: fields[4],
|
||||
})
|
||||
}
|
||||
|
||||
err := sc.Err()
|
||||
if err != nil {
|
||||
fatalf("read %s: %v", name, err)
|
||||
}
|
||||
|
||||
return recs, malformed
|
||||
}
|
||||
|
||||
// dupeGroup is one set of candidate-duplicate files: identical size,
|
||||
// head hash, and tail hash. paths is sorted lexicographically; the
|
||||
// first entry is the group's "first", the rest are dupes.
|
||||
type dupeGroup struct {
|
||||
size int64
|
||||
paths []string
|
||||
}
|
||||
|
||||
// runReport implements the report subcommand: it reads a scan stream
|
||||
// from the named file (or stdin when absent or "-") and prints the
|
||||
// file-level duplicates report as TSV on stdout. It never touches the
|
||||
// scanned filesystem; its only I/O is the scan input, stdout, and
|
||||
// stderr. args holds the positional arguments already validated by
|
||||
// cobra (at most one).
|
||||
func runReport(args []string) {
|
||||
in, name, closer := openScanInput(args)
|
||||
defer closer()
|
||||
|
||||
recs, malformed := parseScanStream(in, name)
|
||||
records := len(recs) + malformed
|
||||
dupes := collectDupeGroups(recs)
|
||||
|
||||
out := bufio.NewWriterSize(os.Stdout, ioBufSize)
|
||||
|
||||
_, err := fmt.Fprintln(out, "first\tdupe\tsize")
|
||||
if err != nil {
|
||||
fatalf("write stdout: %v", err)
|
||||
}
|
||||
|
||||
dupeFiles := 0
|
||||
|
||||
var reclaimable int64
|
||||
|
||||
for _, g := range dupes {
|
||||
for _, p := range g.paths[1:] {
|
||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
||||
g.paths[0], p, g.size)
|
||||
if err != nil {
|
||||
fatalf("write stdout: %v", err)
|
||||
return nil
|
||||
}
|
||||
|
||||
_, writeErr = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
||||
escapePath(first), escapePath(path), size)
|
||||
dupeFiles++
|
||||
reclaimable += g.size
|
||||
reclaimable += size
|
||||
|
||||
return writeErr
|
||||
})
|
||||
|
||||
if writeErr != nil {
|
||||
return fmt.Errorf("write stdout: %w", writeErr)
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
return fmt.Errorf("database %s: %w", dbPath, err)
|
||||
}
|
||||
|
||||
err = out.Flush()
|
||||
if err != nil {
|
||||
fatalf("write stdout: %v", err)
|
||||
return fmt.Errorf("write stdout: %w", err)
|
||||
}
|
||||
|
||||
fmt.Fprintf(os.Stderr,
|
||||
"report: %d records read%s, %d duplicate groups, %d dupe files, %s reclaimable\n",
|
||||
records, malformedNote(malformed), len(dupes), dupeFiles,
|
||||
humanBytes(reclaimable))
|
||||
"report: %d records read, %d duplicate groups, %d dupe files, "+
|
||||
"%s reclaimable\n",
|
||||
records, groups, dupeFiles, humanBytes(reclaimable))
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// collectDupeGroups groups records by signature and returns every group
|
||||
// with two or more paths, each group's paths sorted lexicographically,
|
||||
// groups ordered by size descending then by first path ascending.
|
||||
func collectDupeGroups(recs []scanRec) []dupeGroup {
|
||||
groups := make(map[fileSig][]string)
|
||||
|
||||
for _, r := range recs {
|
||||
k := fileSig{size: r.size, head: r.head, tail: r.tail}
|
||||
groups[k] = append(groups[k], r.path)
|
||||
// escapePath returns a path as it is written in a report column (README
|
||||
// "Report output format"): a backslash, tab, newline or carriage return
|
||||
// becomes \\, \t, \n or \r, and every other byte is kept as it is.
|
||||
// Grouping and sorting use the raw path, never this form.
|
||||
func escapePath(p string) string {
|
||||
// Most paths need no escaping; skip building a replacer for them.
|
||||
if !strings.ContainsAny(p, "\\\t\n\r") {
|
||||
return p
|
||||
}
|
||||
|
||||
var dupes []dupeGroup
|
||||
|
||||
for k, paths := range groups {
|
||||
if len(paths) < minGroupSize {
|
||||
continue
|
||||
}
|
||||
|
||||
slices.Sort(paths)
|
||||
dupes = append(dupes, dupeGroup{size: k.size, paths: paths})
|
||||
}
|
||||
|
||||
// Biggest reclaimable space first; ties broken by first path.
|
||||
slices.SortFunc(dupes, func(a, b dupeGroup) int {
|
||||
if a.size != b.size {
|
||||
if a.size > b.size {
|
||||
return -1
|
||||
}
|
||||
|
||||
return 1
|
||||
}
|
||||
|
||||
return strings.Compare(a.paths[0], b.paths[0])
|
||||
})
|
||||
|
||||
return dupes
|
||||
}
|
||||
|
||||
// malformedNote formats the optional malformed-record note for the
|
||||
// stderr summaries.
|
||||
func malformedNote(malformed int) string {
|
||||
if malformed == 0 {
|
||||
return ""
|
||||
}
|
||||
|
||||
return fmt.Sprintf(" (%d malformed, skipped)", malformed)
|
||||
}
|
||||
|
||||
// splitNUL is a bufio.SplitFunc for NUL-terminated records. Trailing
|
||||
// data without a terminator at EOF is returned as a final record.
|
||||
func splitNUL(data []byte, atEOF bool) (int, []byte, error) {
|
||||
if i := bytes.IndexByte(data, 0); i >= 0 {
|
||||
return i + 1, data[:i], nil
|
||||
}
|
||||
|
||||
if atEOF && len(data) > 0 {
|
||||
return len(data), data, nil
|
||||
}
|
||||
|
||||
return 0, nil, nil
|
||||
return strings.NewReplacer(
|
||||
`\`, `\\`, "\t", `\t`, "\n", `\n`, "\r", `\r`,
|
||||
).Replace(p)
|
||||
}
|
||||
|
||||
// humanBytes formats a byte count in human units (binary prefixes).
|
||||
|
||||
+285
-92
@@ -1,108 +1,269 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// mkRecord serializes one scan record in the on-the-wire format.
|
||||
func mkRecord(size, mtime int64, head, tail, path string) string {
|
||||
return fmt.Sprintf("%d\t%d\t%s\t%s\t%s\x00",
|
||||
size, mtime, head, tail, path)
|
||||
// awkwardDir is a directory name holding every byte the reports escape.
|
||||
const awkwardDir = "/d/\tone\ntwo\rthree\\four"
|
||||
|
||||
// awkwardPairRecs is a duplicate pair in sibling directories /d/A and
|
||||
// awkwardDir. A raw tab sorts before "A" but its escaped form `\t`
|
||||
// sorts after it, so awkwardDir coming first shows that sorting uses
|
||||
// the raw path.
|
||||
func awkwardPairRecs() []scanRec {
|
||||
return []scanRec{
|
||||
{size: 5, head: "h", tail: "t", content: "c", path: "/d/A/f"},
|
||||
{size: 5, head: "h", tail: "t", content: "c", path: awkwardDir + "/f"},
|
||||
}
|
||||
}
|
||||
|
||||
func TestSplitNULScanner(t *testing.T) {
|
||||
t.Parallel()
|
||||
// seedDatabase writes recs into a fresh database and returns its path.
|
||||
func seedDatabase(t *testing.T, recs []scanRec) string {
|
||||
t.Helper()
|
||||
|
||||
sc := bufio.NewScanner(strings.NewReader("a\x00bb\x00\x00tail"))
|
||||
sc.Split(splitNUL)
|
||||
path := testDBPath(t)
|
||||
|
||||
var got []string
|
||||
for sc.Scan() {
|
||||
got = append(got, sc.Text())
|
||||
}
|
||||
|
||||
err := sc.Err()
|
||||
db, err := openScanDatabase(t.Context(), path)
|
||||
if err != nil {
|
||||
t.Fatalf("scanner error: %v", err)
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
want := []string{"a", "bb", "", "tail"}
|
||||
if !slices.Equal(got, want) {
|
||||
t.Fatalf("tokens = %q, want %q", got, want)
|
||||
err = applyChanges(t.Context(), db, recs, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
err = db.Close()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return path
|
||||
}
|
||||
|
||||
// dupeGroup is one duplicate group as report reads it: the size, and
|
||||
// the paths in report order, first path first.
|
||||
type dupeGroup struct {
|
||||
size int64
|
||||
paths []string
|
||||
}
|
||||
|
||||
// dupeGroups returns the duplicate groups report reads from db, in
|
||||
// report order.
|
||||
func dupeGroups(t *testing.T, db *sql.DB) []dupeGroup {
|
||||
t.Helper()
|
||||
|
||||
var groups []dupeGroup
|
||||
|
||||
_, err := loadDupeRows(t.Context(), db,
|
||||
func(first, path string, size int64) error {
|
||||
if path == first {
|
||||
groups = append(groups, dupeGroup{size: size})
|
||||
}
|
||||
|
||||
g := &groups[len(groups)-1]
|
||||
g.paths = append(g.paths, path)
|
||||
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return groups
|
||||
}
|
||||
|
||||
// dupeGroupsOf writes recs into a fresh database and returns the
|
||||
// duplicate groups report reads from it.
|
||||
func dupeGroupsOf(t *testing.T, recs []scanRec) []dupeGroup {
|
||||
t.Helper()
|
||||
|
||||
db := openTestDB(t)
|
||||
|
||||
err := applyChanges(t.Context(), db, recs, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return dupeGroups(t, db)
|
||||
}
|
||||
|
||||
func TestRunReportEscapesPaths(t *testing.T) {
|
||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{cmdReport}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
}
|
||||
|
||||
want := "first\tdupe\tsize\n" +
|
||||
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
|
||||
if got := stdout.String(); got != want {
|
||||
t.Errorf("stdout = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseScanStream(t *testing.T) {
|
||||
func TestReportStdoutFailsWhileReading(t *testing.T) {
|
||||
// Each row holds two paths longer than dir, so the report is more
|
||||
// than twice the stdout buffer and stdout fails while rows are
|
||||
// still being read, not at the final flush.
|
||||
dir := "/" + strings.Repeat("d", 4096)
|
||||
|
||||
recs := make([]scanRec, ioBufSize/len(dir))
|
||||
for i := range recs {
|
||||
recs[i] = scanRec{
|
||||
size: 1, head: "h", tail: "t", content: "c",
|
||||
path: fmt.Sprintf("%s/%d", dir, i),
|
||||
}
|
||||
}
|
||||
|
||||
t.Setenv(databaseEnv, seedDatabase(t, recs))
|
||||
|
||||
err := runReport(t.Context(), failingWriter{})
|
||||
if !errors.Is(err, errWriteFailed) ||
|
||||
!strings.HasPrefix(err.Error(), "write stdout: ") {
|
||||
t.Errorf("error = %v, want write stdout: %v", err, errWriteFailed)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunReportsIgnoreInsertionOrder(t *testing.T) {
|
||||
// README §Constraints: identical database contents give identical
|
||||
// output, whatever order the records were inserted in.
|
||||
recs := append(smokeTreeRecs(), awkwardPairRecs()...)
|
||||
recs = append(recs,
|
||||
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/2"},
|
||||
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/1"},
|
||||
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/2"},
|
||||
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/1"},
|
||||
scanRec{size: 50, path: "/x/unhashed"},
|
||||
)
|
||||
|
||||
reversed := slices.Clone(recs)
|
||||
slices.Reverse(reversed)
|
||||
|
||||
for _, name := range []string{cmdReport, cmdTrees} {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
t.Setenv(databaseEnv, seedDatabase(t, recs))
|
||||
|
||||
forward := runStdout(t, name)
|
||||
|
||||
t.Setenv(databaseEnv, seedDatabase(t, reversed))
|
||||
|
||||
backward := runStdout(t, name)
|
||||
|
||||
if strings.Count(forward, "\n") < 3 {
|
||||
t.Errorf("stdout = %q, want at least two rows", forward)
|
||||
}
|
||||
|
||||
if forward != backward {
|
||||
t.Errorf("stdout depends on insertion order: %q vs %q",
|
||||
forward, backward)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// runStdout runs the subcommand name and returns its stdout, failing
|
||||
// the test unless it succeeds.
|
||||
func runStdout(t *testing.T, name string) string {
|
||||
t.Helper()
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{name}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(%s) = %d, want %d; stderr: %s",
|
||||
name, code, exitOK, stderr.String())
|
||||
}
|
||||
|
||||
return stdout.String()
|
||||
}
|
||||
|
||||
func TestEscapePath(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
stream := mkRecord(10, 1, "h1", "t1", "/a/x") +
|
||||
"garbage-without-tabs\x00" +
|
||||
"notanumber\t1\th\tt\t/a/bad\x00" +
|
||||
mkRecord(20, 2, "h2", "t2", "/a/tab\tin\tname") +
|
||||
"30\t3\th3\tt3\t/trailing/no-nul"
|
||||
|
||||
recs, malformed := parseScanStream(strings.NewReader(stream), "test")
|
||||
|
||||
if malformed != 2 {
|
||||
t.Errorf("malformed = %d, want 2", malformed)
|
||||
cases := map[string]string{
|
||||
"/srv/plain": "/srv/plain",
|
||||
"/a\tb": `/a\tb`,
|
||||
"/a\nb": `/a\nb`,
|
||||
"/a\rb": `/a\rb`,
|
||||
`/a\b`: `/a\\b`,
|
||||
`/a\tb`: `/a\\tb`,
|
||||
"/not-utf8\xff": "/not-utf8\xff",
|
||||
}
|
||||
|
||||
want := []scanRec{
|
||||
{size: 10, head: "h1", tail: "t1", path: "/a/x"},
|
||||
{size: 20, head: "h2", tail: "t2", path: "/a/tab\tin\tname"},
|
||||
{size: 30, head: "h3", tail: "t3", path: "/trailing/no-nul"},
|
||||
for in, want := range cases {
|
||||
if got := escapePath(in); got != want {
|
||||
t.Errorf("escapePath(%q) = %q, want %q", in, got, want)
|
||||
}
|
||||
if !slices.Equal(recs, want) {
|
||||
t.Fatalf("recs = %+v, want %+v", recs, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseScanStreamPathWithNewline(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
in := strings.NewReader(mkRecord(5, 9, "h", "t", "/a/new\nline"))
|
||||
|
||||
recs, malformed := parseScanStream(in, "test")
|
||||
if malformed != 0 || len(recs) != 1 {
|
||||
t.Fatalf("got %d recs, %d malformed, want 1, 0",
|
||||
len(recs), malformed)
|
||||
// TestWarnfEscapes checks that a warning naming a path that holds a
|
||||
// newline is still one line.
|
||||
//
|
||||
//nolint:paralleltest // replaces the process-wide os.Stderr
|
||||
func TestWarnfEscapes(t *testing.T) {
|
||||
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if recs[0].path != "/a/new\nline" {
|
||||
t.Fatalf("path = %q, want %q", recs[0].path, "/a/new\nline")
|
||||
saved := os.Stderr
|
||||
os.Stderr = f
|
||||
|
||||
t.Cleanup(func() {
|
||||
os.Stderr = saved
|
||||
|
||||
_ = f.Close()
|
||||
})
|
||||
|
||||
(&progress{}).warnf("stat %s: %s", "/d/a\nb", "gone")
|
||||
|
||||
_, err = f.Seek(0, io.SeekStart)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
got, err := io.ReadAll(f)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
want := `stat /d/a\nb: gone` + "\n"
|
||||
if string(got) != want {
|
||||
t.Errorf("warning = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseScanStreamEmpty(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
recs, malformed := parseScanStream(strings.NewReader(""), "test")
|
||||
if len(recs) != 0 || malformed != 0 {
|
||||
t.Fatalf("got %d recs, %d malformed, want 0, 0",
|
||||
len(recs), malformed)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroups(t *testing.T) {
|
||||
func TestDupeGroups(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
recs := []scanRec{
|
||||
{size: 100, head: "h", tail: "t", path: "/z/b"},
|
||||
{size: 100, head: "h", tail: "t", path: "/z/a"},
|
||||
{size: 100, head: "h", tail: "t", path: "/z/c"},
|
||||
{size: 4000, head: "H", tail: "T", path: "/big/2"},
|
||||
{size: 4000, head: "H", tail: "T", path: "/big/1"},
|
||||
{size: 100, head: "h", tail: "t", content: "c", path: "/z/b"},
|
||||
{size: 100, head: "h", tail: "t", content: "c", path: "/z/a"},
|
||||
{size: 100, head: "h", tail: "t", content: "c", path: "/z/c"},
|
||||
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/2"},
|
||||
{size: 4000, head: "H", tail: "T", content: "C", path: "/big/1"},
|
||||
// Same size as the /z group but a different head hash.
|
||||
{size: 100, head: "other", tail: "t", path: "/z/d"},
|
||||
{size: 100, head: "other", tail: "t", content: "c", path: "/z/d"},
|
||||
// A singleton signature must not form a group.
|
||||
{size: 7, head: "u", tail: "u", path: "/lonely"},
|
||||
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
groups := dupeGroupsOf(t, recs)
|
||||
if len(groups) != 2 {
|
||||
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
||||
}
|
||||
@@ -120,17 +281,61 @@ func TestCollectDupeGroups(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
||||
func TestDupeGroupsContentSeparates(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// Same size, head, and tail, but different content hashes: the final
|
||||
// rung keeps them apart, so no group forms. Matching content groups.
|
||||
// Records without a content hash never group, not even with each
|
||||
// other.
|
||||
recs := []scanRec{
|
||||
{size: 50, head: "b", tail: "b", path: "/beta/2"},
|
||||
{size: 50, head: "b", tail: "b", path: "/beta/1"},
|
||||
{size: 50, head: "a", tail: "a", path: "/alpha/2"},
|
||||
{size: 50, head: "a", tail: "a", path: "/alpha/1"},
|
||||
{size: 100, head: "h", tail: "t", content: "c1", path: "/a"},
|
||||
{size: 100, head: "h", tail: "t", content: "c2", path: "/b"},
|
||||
{size: 100, head: "h", tail: "t", content: "c1", path: "/c"},
|
||||
{size: 100, head: "h", tail: "t", path: "/d"},
|
||||
{size: 100, head: "h", tail: "t", path: "/e"},
|
||||
}
|
||||
|
||||
groups := collectDupeGroups(recs)
|
||||
groups := dupeGroupsOf(t, recs)
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
||||
len(groups))
|
||||
}
|
||||
|
||||
if !slices.Equal(groups[0].paths, []string{"/a", "/c"}) {
|
||||
t.Errorf("group paths = %q, want /a /c", groups[0].paths)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDupeGroupsMtimeExcluded(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// mtime is informational only; records differing only in mtime
|
||||
// still group together.
|
||||
recs := []scanRec{
|
||||
{size: 9, mtime: 100, head: "h", tail: "t", content: "c", path: "/m/1"},
|
||||
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
||||
}
|
||||
|
||||
groups := dupeGroupsOf(t, recs)
|
||||
if len(groups) != 1 {
|
||||
t.Fatalf("len(groups) = %d, want 1", len(groups))
|
||||
}
|
||||
}
|
||||
|
||||
func TestDupeGroupsTieBreak(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// The hashes sort opposite to the first paths, so ordering the
|
||||
// groups by hash instead of by first path fails this test.
|
||||
recs := []scanRec{
|
||||
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/2"},
|
||||
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/1"},
|
||||
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/2"},
|
||||
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/1"},
|
||||
}
|
||||
|
||||
groups := dupeGroupsOf(t, recs)
|
||||
if len(groups) != 2 {
|
||||
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
||||
}
|
||||
@@ -142,22 +347,22 @@ func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollectDupeGroupsDeterministic(t *testing.T) {
|
||||
func TestDupeGroupsDeterministic(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
recs := []scanRec{
|
||||
{size: 1, head: "a", tail: "a", path: "/p/1"},
|
||||
{size: 1, head: "a", tail: "a", path: "/p/2"},
|
||||
{size: 2, head: "b", tail: "b", path: "/q/1"},
|
||||
{size: 2, head: "b", tail: "b", path: "/q/2"},
|
||||
{size: 1, head: "a", tail: "a", content: "a", path: "/p/1"},
|
||||
{size: 1, head: "a", tail: "a", content: "a", path: "/p/2"},
|
||||
{size: 2, head: "b", tail: "b", content: "b", path: "/q/1"},
|
||||
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
||||
}
|
||||
|
||||
forward := collectDupeGroups(recs)
|
||||
forward := dupeGroupsOf(t, recs)
|
||||
|
||||
reversed := slices.Clone(recs)
|
||||
slices.Reverse(reversed)
|
||||
|
||||
backward := collectDupeGroups(reversed)
|
||||
backward := dupeGroupsOf(t, reversed)
|
||||
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
|
||||
return a.size == b.size && slices.Equal(a.paths, b.paths)
|
||||
}) {
|
||||
@@ -190,15 +395,3 @@ func TestHumanBytes(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestMalformedNote(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
if got := malformedNote(0); got != "" {
|
||||
t.Errorf("malformedNote(0) = %q, want empty", got)
|
||||
}
|
||||
|
||||
if got := malformedNote(3); got != " (3 malformed, skipped)" {
|
||||
t.Errorf("malformedNote(3) = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
+1549
-122
File diff suppressed because it is too large
Load Diff
Executable
+92
@@ -0,0 +1,92 @@
|
||||
#!/bin/sh
|
||||
# script/bootstrap: install all dependencies needed to build and develop
|
||||
# this repo. Idempotent: every install is guarded by a check so already
|
||||
# installed tools are skipped. Base tooling comes from nix, apt, brew,
|
||||
# or apk (detected in that order); assumes nothing is present (not git,
|
||||
# make, or go). Neither the linter nor the Markdown formatter is
|
||||
# installed: golangci-lint (script/lint) and prettier (script/fmt,
|
||||
# script/fmt-check) run via docker only, pinned by hash, so their only
|
||||
# prerequisite is a working docker — which is warned about, not
|
||||
# installed, because everything except linting and formatting works
|
||||
# without it.
|
||||
set -eu
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
|
||||
PKGMGR=""
|
||||
SUDO=""
|
||||
APT_UPDATED=""
|
||||
|
||||
detect_pkgmgr() {
|
||||
[ -n "$PKGMGR" ] && return 0
|
||||
if command -v nix-env >/dev/null 2>&1; then
|
||||
PKGMGR="nix"
|
||||
elif command -v apt-get >/dev/null 2>&1; then
|
||||
PKGMGR="apt"
|
||||
elif command -v brew >/dev/null 2>&1; then
|
||||
PKGMGR="brew"
|
||||
elif command -v apk >/dev/null 2>&1; then
|
||||
PKGMGR="apk"
|
||||
else
|
||||
echo "bootstrap: no supported package manager (nix, apt, brew, apk)" >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ "$PKGMGR" = "apt" ]; then
|
||||
export DEBIAN_FRONTEND=noninteractive
|
||||
if [ "$(id -u)" != "0" ]; then
|
||||
SUDO="sudo"
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
# pkg_install <nix-attr> <apt-pkg> <brew-formula> <apk-pkg>
|
||||
pkg_install() {
|
||||
detect_pkgmgr
|
||||
case "$PKGMGR" in
|
||||
nix) nix-env -iA "nixpkgs.$1" ;;
|
||||
apt)
|
||||
if [ -z "$APT_UPDATED" ]; then
|
||||
$SUDO env DEBIAN_FRONTEND=noninteractive apt-get update
|
||||
APT_UPDATED=1
|
||||
fi
|
||||
$SUDO env DEBIAN_FRONTEND=noninteractive apt-get install -y "$2"
|
||||
;;
|
||||
brew) brew install "$3" ;;
|
||||
apk) apk add --no-cache "$4" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
missing() {
|
||||
! command -v "$1" >/dev/null 2>&1
|
||||
}
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
|
||||
# System tooling, deliberately unpinned: these come from the host
|
||||
# package manager and whatever version it ships is what the host
|
||||
# gets, so a presence check is the right check. The repo pins no
|
||||
# system toolchain versions — the Go language version is governed by
|
||||
# go.mod, and builds that must be reproducible run in the Docker
|
||||
# image, whose base images are pinned by digest.
|
||||
if missing git; then pkg_install git git git git; fi
|
||||
if missing make; then pkg_install gnumake make make make; fi
|
||||
if missing go; then pkg_install go golang go go; fi
|
||||
|
||||
# Linting and Markdown formatting run via docker only, so docker is
|
||||
# their prerequisite rather than something bootstrap installs. Warn,
|
||||
# do not fail: everything except `make lint`, `make fmt` and
|
||||
# `make fmt-check` — and, through them, `make check`, `make docker`
|
||||
# and the pre-commit hook — works without it.
|
||||
if missing docker; then
|
||||
echo "bootstrap: WARNING: docker not found; make lint, make fmt," >&2
|
||||
echo "bootstrap: make fmt-check, make check and make docker" >&2
|
||||
echo "bootstrap: require it." >&2
|
||||
fi
|
||||
|
||||
go mod download
|
||||
|
||||
echo "bootstrap complete"
|
||||
}
|
||||
|
||||
main "$@"
|
||||
Executable
+14
@@ -0,0 +1,14 @@
|
||||
#!/bin/sh
|
||||
# script/check: run all checks (test, lint, fmt-check). Our own
|
||||
# extension to scripts-to-rule-them-all. Must not modify any files.
|
||||
set -eu
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||
|
||||
main() {
|
||||
"$SCRIPT_DIR/test"
|
||||
"$SCRIPT_DIR/lint"
|
||||
"$SCRIPT_DIR/fmt-check"
|
||||
}
|
||||
|
||||
main "$@"
|
||||
Executable
+34
@@ -0,0 +1,34 @@
|
||||
#!/bin/sh
|
||||
# script/cibuild: run the CI build. The Gitea workflow runs this on
|
||||
# push.
|
||||
#
|
||||
# The Dockerfile runs the gates individually as build steps, not the
|
||||
# make check aggregate: the lint stage runs the gofmt check,
|
||||
# script/verify-lint-image-pin, golangci-lint config verify and
|
||||
# golangci-lint run; the markdown stage runs the prettier check; the
|
||||
# build stage, dropped to an unprivileged user, runs make test. None of
|
||||
# make lint, make fmt-check or make check appears, because each runs
|
||||
# docker, and docker cannot run inside a docker build. Nothing is
|
||||
# skipped by that — the linter, gofmt and prettier are invoked directly
|
||||
# in their stages, and the build stage's COPY --from lines make those
|
||||
# stages prerequisites, so BuildKit must finish them first. Between the
|
||||
# three stages everything make check would run has run, which is why a
|
||||
# successful build here implies the repo is green.
|
||||
#
|
||||
# That implication holds only because of CHECK_EPOCH. A COPY layer is
|
||||
# invalidated only by changed content, and a rebuild of an unchanged
|
||||
# checkout sends the same content, so without a fresh value here Docker
|
||||
# serves the gate layers from cache and the build reports a green it
|
||||
# never earned. Passing the current epoch invalidates the gate
|
||||
# layers on every run while leaving the pinned base images and
|
||||
# go mod download cached; see the Dockerfile for the placement.
|
||||
set -eu
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
docker build --build-arg CHECK_EPOCH="$(date +%s)" .
|
||||
}
|
||||
|
||||
main "$@"
|
||||
Executable
+25
@@ -0,0 +1,25 @@
|
||||
#!/bin/sh
|
||||
# script/docker: build the Docker image tagged with the project name.
|
||||
# The tag comes from script/projectname.
|
||||
#
|
||||
# CHECK_EPOCH is passed for the same reason script/cibuild passes it:
|
||||
# without it Docker serves the Dockerfile's gate layers from cache on an
|
||||
# unchanged tree and this exits 0 having run none of the lint stage's
|
||||
# gates, the markdown stage's prettier gate or the builder stage's test
|
||||
# gate. This is the set of gates a developer or reviewer runs by hand,
|
||||
# so a cached pass here is the most misleading result the repo can
|
||||
# produce. Dependency layers sit above the ARG and stay cached.
|
||||
set -eu
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
docker build \
|
||||
--build-arg CHECK_EPOCH="$(date +%s)" \
|
||||
-t "$("$SCRIPT_DIR/projectname")" \
|
||||
.
|
||||
}
|
||||
|
||||
main "$@"
|
||||
Executable
+22
@@ -0,0 +1,22 @@
|
||||
#!/bin/sh
|
||||
# script/fmt: format all files (writes): the Go sources with gofmt, the
|
||||
# Markdown with prettier. prettier is never installed on the host: it
|
||||
# runs from the Dockerfile's prettier stage with the repository mounted,
|
||||
# as the calling user so the files it rewrites keep their owner. The tag
|
||||
# makes each build replace the previous image instead of leaving another
|
||||
# one behind.
|
||||
set -eu
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
gofmt -s -w .
|
||||
image="$("$SCRIPT_DIR/projectname")-prettier"
|
||||
docker build -q --target prettier -t "$image" . >/dev/null
|
||||
docker run --rm --user "$(id -u):$(id -g)" -v "$ROOT:/src" "$image" \
|
||||
prettier --write '**/*.md' --tab-width 4 --prose-wrap always
|
||||
}
|
||||
|
||||
main "$@"
|
||||
Executable
+34
@@ -0,0 +1,34 @@
|
||||
#!/bin/sh
|
||||
# script/fmt-check: check formatting (read-only). Same scope as
|
||||
# script/fmt, but fails instead of writing. gofmt and prettier both run
|
||||
# every time and each reports its own failure, so the output says which
|
||||
# one failed.
|
||||
set -eu
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
status=0
|
||||
|
||||
files="$(gofmt -s -l .)"
|
||||
if [ -n "$files" ]; then
|
||||
echo "gofmt: files not formatted:" >&2
|
||||
echo "$files" >&2
|
||||
status=1
|
||||
fi
|
||||
|
||||
# Same image as script/fmt; see there.
|
||||
image="$("$SCRIPT_DIR/projectname")-prettier"
|
||||
docker build -q --target prettier -t "$image" . >/dev/null
|
||||
if ! docker run --rm -v "$ROOT:/src:ro" "$image" \
|
||||
prettier --check '**/*.md' --tab-width 4 --prose-wrap always; then
|
||||
echo "prettier: Markdown not formatted; run make fmt" >&2
|
||||
status=1
|
||||
fi
|
||||
|
||||
exit "$status"
|
||||
}
|
||||
|
||||
main "$@"
|
||||
Executable
+20
@@ -0,0 +1,20 @@
|
||||
#!/bin/sh
|
||||
# script/install-precommit: install the git pre-commit hook that runs
|
||||
# script/precommit. Our own extension to scripts-to-rule-them-all.
|
||||
# Hooks are shared between the main checkout and all worktrees, so
|
||||
# resolve the common git dir instead of assuming .git is a directory.
|
||||
set -eu
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
hooks_dir="$(git rev-parse --git-common-dir)/hooks"
|
||||
mkdir -p "$hooks_dir"
|
||||
hook="$hooks_dir/pre-commit"
|
||||
printf '#!/bin/sh\nset -e\nscript/precommit\n' > "$hook"
|
||||
chmod +x "$hook"
|
||||
echo "pre-commit hook installed: runs script/precommit"
|
||||
}
|
||||
|
||||
main "$@"
|
||||
Executable
+29
@@ -0,0 +1,29 @@
|
||||
#!/bin/sh
|
||||
# script/lint: run the linter. golangci-lint is never installed on a
|
||||
# host: it runs via docker only, one way, everywhere — this builds
|
||||
# Dockerfile.lint, which COPYs the repo into the digest-pinned
|
||||
# golangci-lint image and lints as a build step, so a successful build
|
||||
# is a clean lint. The only prerequisite is a working docker. The gate
|
||||
# steps make no network calls of their own, but Dockerfile.lint runs
|
||||
# `go mod download` above them, so a cold cache does reach the network
|
||||
# (as does pulling the pinned image); that layer stays cached, and once
|
||||
# it is warm this runs offline until go.mod or go.sum changes.
|
||||
#
|
||||
# CHECK_EPOCH is what makes the result mean anything. Without it docker
|
||||
# serves the gate layers from cache on an unchanged tree and this exits
|
||||
# 0 in well under a second having run no linter. The PID is in the value
|
||||
# as well as the epoch because two lint runs land inside the same second
|
||||
# easily, and `date +%s` alone would cache the second one.
|
||||
set -eu
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
docker build \
|
||||
--build-arg CHECK_EPOCH="$(date +%s)-$$" \
|
||||
-f Dockerfile.lint \
|
||||
.
|
||||
}
|
||||
|
||||
main "$@"
|
||||
Executable
+21
@@ -0,0 +1,21 @@
|
||||
#!/bin/sh
|
||||
# script/precommit: run by the git pre-commit hook; fails the commit if
|
||||
# checks fail. Our own extension to scripts-to-rule-them-all. Go extra:
|
||||
# go mod tidy must be a no-op before the checks run.
|
||||
set -eu
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
go mod tidy
|
||||
if ! git diff --exit-code -- go.mod go.sum; then
|
||||
echo "precommit: go mod tidy changed go.mod/go.sum;" \
|
||||
"stage the changes and retry" >&2
|
||||
exit 1
|
||||
fi
|
||||
"$SCRIPT_DIR/check"
|
||||
}
|
||||
|
||||
main "$@"
|
||||
Executable
+12
@@ -0,0 +1,12 @@
|
||||
#!/bin/sh
|
||||
# script/projectname: output the name of this project. Our own
|
||||
# extension to scripts-to-rule-them-all. Other scripts that need the
|
||||
# name (e.g. script/docker) call this, so they can stay identical
|
||||
# across all repos.
|
||||
set -eu
|
||||
|
||||
main() {
|
||||
echo "sfdupes"
|
||||
}
|
||||
|
||||
main "$@"
|
||||
Executable
+13
@@ -0,0 +1,13 @@
|
||||
#!/bin/sh
|
||||
# script/setup: set up the repo for development after a fresh clone:
|
||||
# installs dependencies (script/bootstrap) and the git pre-commit hook.
|
||||
set -eu
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||
|
||||
main() {
|
||||
"$SCRIPT_DIR/bootstrap"
|
||||
"$SCRIPT_DIR/install-precommit"
|
||||
}
|
||||
|
||||
main "$@"
|
||||
Executable
+17
@@ -0,0 +1,17 @@
|
||||
#!/bin/sh
|
||||
# script/test: run the test suite. Reruns verbosely on failure so CI
|
||||
# logs show which test failed.
|
||||
set -eu
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
go test -timeout 30s -cover ./... || {
|
||||
echo "--- Rerunning with -v for details ---"
|
||||
go test -timeout 30s -v ./...
|
||||
exit 1
|
||||
}
|
||||
}
|
||||
|
||||
main "$@"
|
||||
Executable
+84
@@ -0,0 +1,84 @@
|
||||
#!/bin/sh
|
||||
# script/verify-lint-image-pin: fail unless the golangci-lint image
|
||||
# referenced by Dockerfile.lint and the one referenced by the main
|
||||
# Dockerfile's lint stage are the same image at the same digest. Our own
|
||||
# extension to scripts-to-rule-them-all, not one of its entrypoints.
|
||||
#
|
||||
# The linter version is pinned in two independent files. That is the
|
||||
# shape #42 turned into a build failure rather than tolerate: nothing
|
||||
# else keeps the two in sync, and a bump applied to one file alone would
|
||||
# leave `make lint` and the fail-fast lint stage of `make docker`
|
||||
# linting the same tree against different rulesets, both green. This is
|
||||
# the single guard that stops it, run as a gate in both files.
|
||||
#
|
||||
# It deliberately restates neither pin. A hardcoded expected digest here
|
||||
# would be a third copy — one more thing to bump, and the same drift one
|
||||
# file further out. It compares the two files to each other and knows
|
||||
# nothing about which version is correct.
|
||||
#
|
||||
# A reference that cannot be read is a hard failure, not a skip: a
|
||||
# comparison of two empty strings succeeds, which would turn this guard
|
||||
# into exactly the unearned green it exists to prevent.
|
||||
set -eu
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
|
||||
LINT_DOCKERFILE="Dockerfile.lint"
|
||||
MAIN_DOCKERFILE="Dockerfile"
|
||||
|
||||
# Echo the single golangci-lint image reference in the named Dockerfile.
|
||||
# Scans every argument of every FROM instruction rather than assuming a
|
||||
# field position, so `FROM --platform=... img AS stage` reads correctly.
|
||||
# Exits non-zero, with a diagnosis, unless there is exactly one.
|
||||
lint_image_ref() {
|
||||
file="$1"
|
||||
|
||||
if [ ! -f "$file" ]; then
|
||||
echo "verify-lint-image-pin: $file: not found" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
refs="$(
|
||||
awk '
|
||||
toupper($1) == "FROM" {
|
||||
for (i = 2; i <= NF; i++) {
|
||||
if ($i ~ /^golangci\/golangci-lint[:@]/) {
|
||||
print $i
|
||||
}
|
||||
}
|
||||
}
|
||||
' "$file"
|
||||
)"
|
||||
|
||||
count="$(printf '%s' "$refs" | grep -c . || true)"
|
||||
if [ "$count" -ne 1 ]; then
|
||||
echo "verify-lint-image-pin: $file: expected exactly one" \
|
||||
"golangci/golangci-lint FROM reference, found $count" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
printf '%s\n' "$refs"
|
||||
}
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
|
||||
lint_ref="$(lint_image_ref "$LINT_DOCKERFILE")"
|
||||
main_ref="$(lint_image_ref "$MAIN_DOCKERFILE")"
|
||||
|
||||
if [ "$lint_ref" != "$main_ref" ]; then
|
||||
echo "verify-lint-image-pin: the linter image is pinned twice and" \
|
||||
"the two pins disagree:" >&2
|
||||
echo "verify-lint-image-pin: $LINT_DOCKERFILE: $lint_ref" >&2
|
||||
echo "verify-lint-image-pin: $MAIN_DOCKERFILE: $main_ref" >&2
|
||||
echo "verify-lint-image-pin: bump both FROM lines together so" \
|
||||
"script/lint and the Dockerfile lint stage keep running the" \
|
||||
"same linter" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "verify-lint-image-pin: $LINT_DOCKERFILE and $MAIN_DOCKERFILE" \
|
||||
"agree on $lint_ref"
|
||||
}
|
||||
|
||||
main "$@"
|
||||
@@ -2,56 +2,66 @@ package main
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"slices"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// fileSig is a file's duplicate signature; mtime is excluded.
|
||||
type fileSig struct {
|
||||
size int64
|
||||
head string
|
||||
tail string
|
||||
}
|
||||
|
||||
// treeNode is one directory reconstructed from the scan stream.
|
||||
// treeNode is one directory reconstructed from the record paths.
|
||||
type treeNode struct {
|
||||
path string
|
||||
parent *treeNode
|
||||
dirs map[string]*treeNode
|
||||
files map[string]fileSig
|
||||
// entries holds the serialized child entries until the digest is
|
||||
// computed from them, and is then dropped.
|
||||
entries []string
|
||||
digest [sha256.Size]byte
|
||||
fileCount int64
|
||||
totalSize int64
|
||||
}
|
||||
|
||||
// runTrees implements the trees subcommand: it reads a scan stream from
|
||||
// the named file (or stdin when absent or "-"), reconstructs the
|
||||
// directory hierarchy from the record paths, computes a Merkle-style
|
||||
// digest per directory, and prints maximal duplicate-tree groups as TSV
|
||||
// on stdout. It never touches the scanned filesystem; its only I/O is
|
||||
// the scan input, stdout, and stderr. args holds the positional
|
||||
// arguments already validated by cobra (at most one).
|
||||
func runTrees(args []string) {
|
||||
in, name, closer := openScanInput(args)
|
||||
defer closer()
|
||||
// runTrees implements the trees subcommand: it reads every record from
|
||||
// the database in path order, reconstructs the directory hierarchy from
|
||||
// the record paths, computes a Merkle-style digest per directory, and
|
||||
// prints maximal duplicate-tree groups as TSV on stdout. It never
|
||||
// touches the scanned filesystem; its only I/O is the database, stdout,
|
||||
// and stderr. Any database problem, including a missing database, is
|
||||
// fatal.
|
||||
func runTrees(ctx context.Context, stdout io.Writer) error {
|
||||
dbPath := databasePath()
|
||||
|
||||
recs, malformed := parseScanStream(in, name)
|
||||
records := len(recs) + malformed
|
||||
db, err := openReportDatabase(ctx, dbPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
super, allDirs := buildHierarchy(recs)
|
||||
super.compute()
|
||||
defer func() { _ = db.Close() }()
|
||||
|
||||
records := 0
|
||||
tree := newTreeBuilder()
|
||||
|
||||
err = loadFileRows(ctx, db, func(r scanRec) {
|
||||
records++
|
||||
|
||||
tree.add(r)
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("database %s: %w", dbPath, err)
|
||||
}
|
||||
|
||||
super, allDirs := tree.finish()
|
||||
|
||||
dupes := collectTreeGroups(allDirs, super)
|
||||
|
||||
out := bufio.NewWriterSize(os.Stdout, ioBufSize)
|
||||
out := bufio.NewWriterSize(stdout, ioBufSize)
|
||||
|
||||
_, err := fmt.Fprintln(out, "first\tdupe\tfiles\tsize")
|
||||
_, err = fmt.Fprintln(out, "first\tdupe\tfiles\tsize")
|
||||
if err != nil {
|
||||
fatalf("write stdout: %v", err)
|
||||
return fmt.Errorf("write stdout: %w", err)
|
||||
}
|
||||
|
||||
dupeTrees := 0
|
||||
@@ -62,9 +72,10 @@ func runTrees(args []string) {
|
||||
first := g[0]
|
||||
for _, n := range g[1:] {
|
||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\t%d\n",
|
||||
first.path, n.path, first.fileCount, first.totalSize)
|
||||
escapePath(first.path), escapePath(n.path),
|
||||
first.fileCount, first.totalSize)
|
||||
if err != nil {
|
||||
fatalf("write stdout: %v", err)
|
||||
return fmt.Errorf("write stdout: %w", err)
|
||||
}
|
||||
|
||||
dupeTrees++
|
||||
@@ -74,59 +85,134 @@ func runTrees(args []string) {
|
||||
|
||||
err = out.Flush()
|
||||
if err != nil {
|
||||
fatalf("write stdout: %v", err)
|
||||
return fmt.Errorf("write stdout: %w", err)
|
||||
}
|
||||
|
||||
fmt.Fprintf(os.Stderr,
|
||||
"trees: %d records read%s, %d duplicate tree groups, %d dupe trees, %s reclaimable\n",
|
||||
records, malformedNote(malformed), len(dupes), dupeTrees,
|
||||
humanBytes(reclaimable))
|
||||
"trees: %d records read, %d duplicate tree groups, %d dupe trees, "+
|
||||
"%s reclaimable\n",
|
||||
records, len(dupes), dupeTrees, humanBytes(reclaimable))
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// buildHierarchy reconstructs the directory hierarchy from the record
|
||||
// paths under a synthetic super-root. Paths are split on "/"; for
|
||||
// absolute paths the first component is empty, which simply becomes a
|
||||
// top-level node representing "/". It returns the super-root and every
|
||||
// directory node created.
|
||||
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
||||
// treeBuilder reconstructs the directory hierarchy from records added
|
||||
// in path order, under a synthetic super-root. Paths are split on "/";
|
||||
// for absolute paths the first component is empty, which becomes the
|
||||
// top-level directory with path "/". In path order all the paths under
|
||||
// one directory come together, so a directory is complete once a path
|
||||
// outside it is added: its digest is computed then and its entries are
|
||||
// dropped. Only the directories holding the latest path keep entries.
|
||||
type treeBuilder struct {
|
||||
super *treeNode
|
||||
// open lists the directories holding the latest path, outermost
|
||||
// first, starting with the super-root; names[i] is open[i]'s name.
|
||||
open []*treeNode
|
||||
names []string
|
||||
// dirs lists every completed directory.
|
||||
dirs []*treeNode
|
||||
}
|
||||
|
||||
func newTreeBuilder() *treeBuilder {
|
||||
super := &treeNode{}
|
||||
|
||||
var allDirs []*treeNode
|
||||
return &treeBuilder{
|
||||
super: super,
|
||||
open: []*treeNode{super},
|
||||
names: []string{""},
|
||||
}
|
||||
}
|
||||
|
||||
for _, r := range recs {
|
||||
// add adds one record. Each record must come after the previous one in
|
||||
// path order (byte order); otherwise a completed directory would be
|
||||
// started again as a second directory with the same path.
|
||||
func (b *treeBuilder) add(r scanRec) {
|
||||
comps := strings.Split(r.path, "/")
|
||||
dirNames, name := comps[:len(comps)-1], comps[len(comps)-1]
|
||||
|
||||
node := super
|
||||
for _, c := range comps[:len(comps)-1] {
|
||||
child := node.dirs[c]
|
||||
if child == nil {
|
||||
childPath := c
|
||||
if node != super {
|
||||
childPath = node.path + "/" + c
|
||||
// Keep the open directories that hold this path; complete the rest.
|
||||
depth := 1
|
||||
for depth < len(b.open) && depth <= len(dirNames) &&
|
||||
b.names[depth] == dirNames[depth-1] {
|
||||
depth++
|
||||
}
|
||||
|
||||
child = &treeNode{path: childPath, parent: node}
|
||||
if node.dirs == nil {
|
||||
node.dirs = make(map[string]*treeNode)
|
||||
b.closeTo(depth)
|
||||
|
||||
for _, c := range dirNames[depth-1:] {
|
||||
b.openDir(c)
|
||||
}
|
||||
|
||||
node.dirs[c] = child
|
||||
allDirs = append(allDirs, child)
|
||||
dir := b.open[len(b.open)-1]
|
||||
dir.entries = append(dir.entries, fileEntry(name, r))
|
||||
dir.fileCount++
|
||||
dir.totalSize += r.size
|
||||
}
|
||||
|
||||
node = child
|
||||
// openDir starts the directory called name inside the innermost open
|
||||
// one.
|
||||
func (b *treeBuilder) openDir(name string) {
|
||||
parent := b.open[len(b.open)-1]
|
||||
path := parent.path + "/" + name
|
||||
|
||||
// The root directory's path is "/", not empty, and its children's
|
||||
// paths start with one slash, not two.
|
||||
switch {
|
||||
case parent == b.super && name == "":
|
||||
path = "/"
|
||||
case parent == b.super:
|
||||
path = name
|
||||
case parent.path == "/":
|
||||
path = "/" + name
|
||||
}
|
||||
|
||||
if node.files == nil {
|
||||
node.files = make(map[string]fileSig)
|
||||
b.open = append(b.open, &treeNode{path: path, parent: parent})
|
||||
b.names = append(b.names, name)
|
||||
}
|
||||
|
||||
node.files[comps[len(comps)-1]] = fileSig{
|
||||
size: r.size, head: r.head, tail: r.tail,
|
||||
// closeTo completes the open directories after the first n, innermost
|
||||
// first: each one's digest is computed and entered in its parent along
|
||||
// with its totals.
|
||||
func (b *treeBuilder) closeTo(n int) {
|
||||
for len(b.open) > n {
|
||||
last := len(b.open) - 1
|
||||
dir, name := b.open[last], b.names[last]
|
||||
b.open, b.names = b.open[:last], b.names[:last]
|
||||
|
||||
dir.computeDigest()
|
||||
|
||||
dir.parent.entries = append(dir.parent.entries,
|
||||
"d\x00"+name+"\x00"+string(dir.digest[:]))
|
||||
dir.parent.fileCount += dir.fileCount
|
||||
dir.parent.totalSize += dir.totalSize
|
||||
b.dirs = append(b.dirs, dir)
|
||||
}
|
||||
}
|
||||
|
||||
return super, allDirs
|
||||
// finish completes every open directory and returns the super-root and
|
||||
// every directory.
|
||||
func (b *treeBuilder) finish() (*treeNode, []*treeNode) {
|
||||
b.closeTo(1)
|
||||
|
||||
return b.super, b.dirs
|
||||
}
|
||||
|
||||
// fileEntry serializes a file child for its directory's digest: its
|
||||
// name and its signature (size, head, tail, content); mtime is
|
||||
// excluded.
|
||||
func fileEntry(name string, r scanRec) string {
|
||||
content := r.content
|
||||
|
||||
// A record without a content hash has unknown content (README
|
||||
// "Database"): give it a signature no other file can share, so
|
||||
// trees containing it never compare equal. Real hashes are hex, so
|
||||
// the NUL-prefixed form cannot collide.
|
||||
if content == "" {
|
||||
content = "unhashed\x00" + r.path
|
||||
}
|
||||
|
||||
return "f\x00" + name + "\x00" + strconv.FormatInt(r.size, 10) +
|
||||
"\x00" + r.head + "\x00" + r.tail + "\x00" + content
|
||||
}
|
||||
|
||||
// collectTreeGroups groups directories by digest and returns every
|
||||
@@ -168,38 +254,22 @@ func collectTreeGroups(allDirs []*treeNode, super *treeNode) [][]*treeNode {
|
||||
return dupes
|
||||
}
|
||||
|
||||
// compute fills in digest, fileCount, and totalSize for n and all of
|
||||
// its descendants. A directory's digest is the SHA-256 of its child
|
||||
// entries — files serialized with name and signature, subdirectories
|
||||
// with name and recursive digest — sorted byte-lexicographically.
|
||||
// Filenames cannot contain NUL or "/", so NUL delimiters are
|
||||
// unambiguous.
|
||||
func (n *treeNode) compute() {
|
||||
entries := make([]string, 0, len(n.dirs)+len(n.files))
|
||||
for name, sig := range n.files {
|
||||
entries = append(entries,
|
||||
"f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+
|
||||
"\x00"+sig.head+"\x00"+sig.tail)
|
||||
n.fileCount++
|
||||
n.totalSize += sig.size
|
||||
}
|
||||
|
||||
for name, child := range n.dirs {
|
||||
child.compute()
|
||||
entries = append(entries, "d\x00"+name+"\x00"+string(child.digest[:]))
|
||||
n.fileCount += child.fileCount
|
||||
n.totalSize += child.totalSize
|
||||
}
|
||||
|
||||
slices.Sort(entries)
|
||||
// computeDigest sets n's digest and drops its entries. A directory's
|
||||
// digest is the SHA-256 of its child entries — files serialized with
|
||||
// name and signature, subdirectories with name and recursive digest —
|
||||
// sorted byte-lexicographically. Filenames cannot contain NUL or "/",
|
||||
// so NUL delimiters are unambiguous.
|
||||
func (n *treeNode) computeDigest() {
|
||||
slices.Sort(n.entries)
|
||||
|
||||
h := sha256.New()
|
||||
for _, e := range entries {
|
||||
for _, e := range n.entries {
|
||||
h.Write([]byte(e))
|
||||
h.Write([]byte{0})
|
||||
}
|
||||
|
||||
copy(n.digest[:], h.Sum(nil))
|
||||
n.entries = nil
|
||||
}
|
||||
|
||||
// suppressed reports whether a duplicate-tree group is non-maximal: its
|
||||
|
||||
+145
-30
@@ -1,6 +1,8 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"database/sql"
|
||||
"slices"
|
||||
"testing"
|
||||
)
|
||||
@@ -9,23 +11,56 @@ import (
|
||||
const (
|
||||
f1Head = "f1h"
|
||||
f1Tail = "f1t"
|
||||
f1Content = "f1c"
|
||||
f2Head = "f2h"
|
||||
f2Tail = "f2t"
|
||||
f2Content = "f2c"
|
||||
)
|
||||
|
||||
// smokeTreeRecs mirrors the README smoke-test tree layout: /d/t1 and
|
||||
// /d/t2 are identical, /d/t3 differs from them only by one filename.
|
||||
func smokeTreeRecs() []scanRec {
|
||||
return []scanRec{
|
||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t1/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t1/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t2/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t2/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, path: "/d/t3/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, path: "/d/t3/sub/f2renamed"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t1/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t1/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t2/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, content: f2Content, path: "/d/t2/sub/f2"},
|
||||
{size: 3000, head: f1Head, tail: f1Tail, content: f1Content, path: "/d/t3/f1"},
|
||||
{size: 100, head: f2Head, tail: f2Tail, content: f2Content,
|
||||
path: "/d/t3/sub/f2renamed"},
|
||||
}
|
||||
}
|
||||
|
||||
// dbTree builds the directory hierarchy from the records in db the way
|
||||
// trees does, and returns the super-root and every directory.
|
||||
func dbTree(t *testing.T, db *sql.DB) (*treeNode, []*treeNode) {
|
||||
t.Helper()
|
||||
|
||||
tree := newTreeBuilder()
|
||||
|
||||
err := loadFileRows(t.Context(), db, tree.add)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return tree.finish()
|
||||
}
|
||||
|
||||
// treeOf writes recs into a fresh database and builds the directory
|
||||
// hierarchy from it the way trees does.
|
||||
func treeOf(t *testing.T, recs []scanRec) (*treeNode, []*treeNode) {
|
||||
t.Helper()
|
||||
|
||||
db := openTestDB(t)
|
||||
|
||||
err := applyChanges(t.Context(), db, recs, nil, nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
return dbTree(t, db)
|
||||
}
|
||||
|
||||
// nodeByPath finds the directory node with the given path.
|
||||
func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode {
|
||||
t.Helper()
|
||||
@@ -56,11 +91,10 @@ func groupPaths(groups [][]*treeNode) [][]string {
|
||||
return out
|
||||
}
|
||||
|
||||
func TestBuildHierarchyCounts(t *testing.T) {
|
||||
func TestTreeCounts(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
super, dirs := buildHierarchy(smokeTreeRecs())
|
||||
super.compute()
|
||||
_, dirs := treeOf(t, smokeTreeRecs())
|
||||
|
||||
d := nodeByPath(t, dirs, "/d")
|
||||
if d.fileCount != 6 || d.totalSize != 9300 {
|
||||
@@ -81,11 +115,98 @@ func TestBuildHierarchyCounts(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestTreeRootPath(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// The root directory's path is "/", never empty, and its
|
||||
// children's paths start with a single slash.
|
||||
_, dirs := treeOf(t, []scanRec{{path: "/f"}, {path: "/srv/g"}})
|
||||
|
||||
got := make([]string, 0, len(dirs))
|
||||
for _, d := range dirs {
|
||||
got = append(got, d.path)
|
||||
}
|
||||
|
||||
slices.Sort(got)
|
||||
|
||||
want := []string{"/", "/srv"}
|
||||
if !slices.Equal(got, want) {
|
||||
t.Fatalf("directory paths = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTreeNamesSortingBeforeSlash(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// In path order "/a/b-x/f" and "/a/b.txt" come between the file
|
||||
// "/a/b" and "/a/b/f", because "-" and "." sort before "/". Each
|
||||
// directory must still be built once, whole, so /a matches /c.
|
||||
recs := make([]scanRec, 0, 8)
|
||||
|
||||
for _, top := range []string{"/a", "/c"} {
|
||||
for _, p := range []string{"/b", "/b-x/f", "/b.txt", "/b/f"} {
|
||||
content := "c"
|
||||
if p == "/b-x/f" {
|
||||
content = "other"
|
||||
}
|
||||
|
||||
recs = append(recs, scanRec{
|
||||
size: 1, head: "h", tail: "t", content: content, path: top + p,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
super, dirs := treeOf(t, recs)
|
||||
|
||||
got := make([]string, 0, len(dirs))
|
||||
for _, d := range dirs {
|
||||
got = append(got, d.path)
|
||||
}
|
||||
|
||||
slices.Sort(got)
|
||||
|
||||
want := []string{"/", "/a", "/a/b", "/a/b-x", "/c", "/c/b", "/c/b-x"}
|
||||
if !slices.Equal(got, want) {
|
||||
t.Fatalf("directory paths = %q, want %q", got, want)
|
||||
}
|
||||
|
||||
groups := collectTreeGroups(dirs, super)
|
||||
|
||||
gotGroups := groupPaths(groups)
|
||||
wantGroups := [][]string{{"/a", "/c"}}
|
||||
|
||||
if !slices.EqualFunc(gotGroups, wantGroups, slices.Equal) {
|
||||
t.Fatalf("groups = %v, want %v", gotGroups, wantGroups)
|
||||
}
|
||||
|
||||
if groups[0][0].fileCount != 4 || groups[0][0].totalSize != 4 {
|
||||
t.Errorf("group totals: %d files %d bytes, want 4 4",
|
||||
groups[0][0].fileCount, groups[0][0].totalSize)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunTreesEscapesPaths(t *testing.T) {
|
||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
||||
|
||||
var stdout, stderr bytes.Buffer
|
||||
|
||||
code := run([]string{cmdTrees}, &stdout, &stderr)
|
||||
if code != exitOK {
|
||||
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
||||
code, exitOK, stderr.String())
|
||||
}
|
||||
|
||||
want := "first\tdupe\tfiles\tsize\n" +
|
||||
`/d/\tone\ntwo\rthree\\four` + "\t/d/A\t1\t5\n"
|
||||
if got := stdout.String(); got != want {
|
||||
t.Errorf("stdout = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTreeDigests(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
super, dirs := buildHierarchy(smokeTreeRecs())
|
||||
super.compute()
|
||||
_, dirs := treeOf(t, smokeTreeRecs())
|
||||
|
||||
t1 := nodeByPath(t, dirs, "/d/t1")
|
||||
t2 := nodeByPath(t, dirs, "/d/t2")
|
||||
@@ -114,12 +235,11 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
|
||||
const sharedTail = "same"
|
||||
|
||||
recs := []scanRec{
|
||||
{size: 10, head: sharedTail, tail: sharedTail, path: "/r/a/f"},
|
||||
{size: 10, head: "DIFF", tail: sharedTail, path: "/r/b/f"},
|
||||
{size: 10, head: sharedTail, tail: sharedTail, content: "c", path: "/r/a/f"},
|
||||
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
|
||||
}
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
super.compute()
|
||||
_, dirs := treeOf(t, recs)
|
||||
|
||||
a := nodeByPath(t, dirs, "/r/a")
|
||||
b := nodeByPath(t, dirs, "/r/b")
|
||||
@@ -132,8 +252,7 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
|
||||
func TestCollectTreeGroupsMaximal(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
super, dirs := buildHierarchy(smokeTreeRecs())
|
||||
super.compute()
|
||||
super, dirs := treeOf(t, smokeTreeRecs())
|
||||
|
||||
groups := collectTreeGroups(dirs, super)
|
||||
|
||||
@@ -157,16 +276,14 @@ func TestCollectTreeGroupsDeterministic(t *testing.T) {
|
||||
|
||||
recs := smokeTreeRecs()
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
super.compute()
|
||||
super, dirs := treeOf(t, recs)
|
||||
|
||||
forward := groupPaths(collectTreeGroups(dirs, super))
|
||||
|
||||
reversed := slices.Clone(recs)
|
||||
slices.Reverse(reversed)
|
||||
|
||||
superR, dirsR := buildHierarchy(reversed)
|
||||
superR.compute()
|
||||
superR, dirsR := treeOf(t, reversed)
|
||||
|
||||
backward := groupPaths(collectTreeGroups(dirsR, superR))
|
||||
if !slices.EqualFunc(forward, backward, slices.Equal) {
|
||||
@@ -181,12 +298,11 @@ func TestCollectTreeGroupsSiblings(t *testing.T) {
|
||||
// Identical sibling dirs share a parent, so their group cannot be
|
||||
// implied by a parent group and must be reported.
|
||||
recs := []scanRec{
|
||||
{size: 10, head: "h", tail: "t", path: "/p/x1/f"},
|
||||
{size: 10, head: "h", tail: "t", path: "/p/x2/f"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x1/f"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
|
||||
}
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
super.compute()
|
||||
super, dirs := treeOf(t, recs)
|
||||
|
||||
got := groupPaths(collectTreeGroups(dirs, super))
|
||||
|
||||
@@ -203,13 +319,12 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) {
|
||||
// extra file, so the parents' digests differ and the x group must
|
||||
// be reported.
|
||||
recs := []scanRec{
|
||||
{size: 10, head: "h", tail: "t", path: "/p/a/x/f"},
|
||||
{size: 99, head: "e", tail: "e", path: "/p/a/extra"},
|
||||
{size: 10, head: "h", tail: "t", path: "/q/b/x/f"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/a/x/f"},
|
||||
{size: 99, head: "e", tail: "e", content: "e", path: "/p/a/extra"},
|
||||
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
|
||||
}
|
||||
|
||||
super, dirs := buildHierarchy(recs)
|
||||
super.compute()
|
||||
super, dirs := treeOf(t, recs)
|
||||
|
||||
got := groupPaths(collectTreeGroups(dirs, super))
|
||||
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
# THIS IS AN AUTOGENERATED FILE. DO NOT EDIT THIS FILE DIRECTLY.
|
||||
# yarn lockfile v1
|
||||
|
||||
|
||||
prettier@3.8.1:
|
||||
version "3.8.1"
|
||||
resolved "https://registry.yarnpkg.com/prettier/-/prettier-3.8.1.tgz#edf48977cf991558f4fcbd8a3ba6015ba2a3a173"
|
||||
integrity sha512-UOnG6LftzbdaHZcKoPFtOcCKztrQ57WkHDeRD9t/PTQtmT0NHSeWWepj6pS0z/N7+08BHFDQVUrfmfMRcZwbMg==
|
||||
Reference in New Issue
Block a user