Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c2804d088b |
@@ -0,0 +1,5 @@
|
|||||||
|
{
|
||||||
|
"worktree": {
|
||||||
|
"bgIsolation": "none"
|
||||||
|
}
|
||||||
|
}
|
||||||
+6
-78
@@ -1,79 +1,7 @@
|
|||||||
# .dockerignore does NOT use .gitignore semantics. Docker matches with
|
.git
|
||||||
# moby/patternmatcher: filepath.Match plus `**`, so `*` does not cross
|
|
||||||
# `/` and an unprefixed pattern is anchored at the context root. Every
|
|
||||||
# depth-independent pattern therefore needs `**/`, or `config/.env` and
|
|
||||||
# `certs/server.key` still ship while this file reads as solved. Only
|
|
||||||
# genuinely root-anchored entries go unprefixed. Never transplant these
|
|
||||||
# into .gitignore, where `**/` is wrong.
|
|
||||||
#
|
|
||||||
# Matching is case-sensitive, so secrets use character ranges rather
|
|
||||||
# than an ALL-CAPS twin, which would still miss `Server.Key`.
|
|
||||||
#
|
|
||||||
# Extend with this repo's own host-built artifacts, written anchored:
|
|
||||||
# `/myapp`, never `**/myapp`, which also matches `cmd/myapp/` and
|
|
||||||
# deletes the package directory from the context.
|
|
||||||
|
|
||||||
# .git is sent without its config. Without a VERSION build argument the
|
|
||||||
# stage that compiles runs `git describe --tags --always` on .git, which
|
|
||||||
# does not need .git/config; that file can hold a credential, such as a
|
|
||||||
# password in a remote URL or the token the CI checkout step stores there.
|
|
||||||
# Each submodule keeps a config with the same exposure in its git directory
|
|
||||||
# under .git/modules/, nested again for a submodule's own submodules, or in
|
|
||||||
# its own .git directory when it keeps one.
|
|
||||||
# KNOWN GAP: a submodule whose name has a `config` segment (`config`,
|
|
||||||
# `deploy/config`, `config/lib`) loses its whole git directory, because
|
|
||||||
# `**/.git/modules/**/config` also matches that segment's directory
|
|
||||||
# under .git/modules/. Go's version stamping then fails the build;
|
|
||||||
# nothing leaks. Name such a submodule without that segment:
|
|
||||||
# `git submodule add --name`.
|
|
||||||
**/.git/config
|
|
||||||
**/.git/modules/**/config
|
|
||||||
|
|
||||||
# Agent scratch: one full checkout of the repo per in-flight agent.
|
|
||||||
# Anchored because it occurs once where agents run at the repo root.
|
|
||||||
# KNOWN GAP: a repo running agents in subdirectories still ships
|
|
||||||
# `services/api/.claude/` and must add its own anchored entry.
|
|
||||||
.claude
|
.claude
|
||||||
|
.DS_Store
|
||||||
# Environment files. `*.env` covers bare `.env` and the `prod.env`
|
sfdupes
|
||||||
# convention. Re-include a committed template with a negation if the
|
*.log
|
||||||
# build needs one: `!docs/example.env`.
|
*.out
|
||||||
**/*.[eE][nN][vV]
|
*.test
|
||||||
**/.[eE][nN][vV].*
|
|
||||||
**/.[eE][nN][vV][rR][cC]
|
|
||||||
|
|
||||||
# Private keys and the bundles carrying them. Public certificates
|
|
||||||
# (*.crt, *.cer) are deliberately absent: they are legitimate inputs.
|
|
||||||
**/*.[pP][eE][mM]
|
|
||||||
**/*.[kK][eE][yY]
|
|
||||||
**/*.[pP]12
|
|
||||||
**/*.[pP][fF][xX]
|
|
||||||
**/[iI][dD]_[rR][sS][aA]
|
|
||||||
**/[iI][dD]_[dD][sS][aA]
|
|
||||||
**/[iI][dD]_[eE][cC][dD][sS][aA]
|
|
||||||
**/[iI][dD]_[eE][cC][dD][sS][aA]_[sS][kK]
|
|
||||||
**/[iI][dD]_[eE][dD]25519
|
|
||||||
**/[iI][dD]_[eE][dD]25519_[sS][kK]
|
|
||||||
|
|
||||||
# Dependencies: restored inside the image, never copied in.
|
|
||||||
**/node_modules
|
|
||||||
|
|
||||||
# OS metadata.
|
|
||||||
**/.DS_Store
|
|
||||||
**/Thumbs.db
|
|
||||||
|
|
||||||
# Editor state: never a build input, and it churns COPY.
|
|
||||||
**/*.swp
|
|
||||||
**/*.swo
|
|
||||||
**/*~
|
|
||||||
**/*.bak
|
|
||||||
**/.idea
|
|
||||||
**/.vscode
|
|
||||||
**/*.sublime-*
|
|
||||||
|
|
||||||
# This repository's host-built artifacts: the binary `make build` writes,
|
|
||||||
# and test binaries, coverage output and logs.
|
|
||||||
/sfdupes
|
|
||||||
/*.test
|
|
||||||
/*.out
|
|
||||||
/*.log
|
|
||||||
|
|||||||
@@ -13,6 +13,3 @@ indent_style = tab
|
|||||||
|
|
||||||
[*.go]
|
[*.go]
|
||||||
indent_style = tab
|
indent_style = tab
|
||||||
|
|
||||||
# This repository's own sections, such as one for another language it
|
|
||||||
# uses, go below this comment, and a re-vendor keeps them.
|
|
||||||
|
|||||||
@@ -1,20 +1,9 @@
|
|||||||
name: check
|
name: check
|
||||||
on: [push]
|
on: [push]
|
||||||
# Free the shared runner: a new push cancels only the same branch's older run.
|
|
||||||
concurrency:
|
|
||||||
group: ${{ github.workflow }}-${{ github.ref }}
|
|
||||||
cancel-in-progress: true
|
|
||||||
jobs:
|
jobs:
|
||||||
check:
|
check:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
# Free the shared runner from a hung build.
|
|
||||||
timeout-minutes: 20
|
|
||||||
steps:
|
steps:
|
||||||
# actions/checkout v4.2.2, 2026-02-22
|
# actions/checkout v4.2.2, 2026-02-22
|
||||||
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
|
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
|
||||||
# script/cibuild needs no token, so none is left in .git/config.
|
|
||||||
with:
|
|
||||||
persist-credentials: false
|
|
||||||
# All history and tags, so git describe finds the version tag.
|
|
||||||
fetch-depth: 0
|
|
||||||
- run: script/cibuild
|
- run: script/cibuild
|
||||||
|
|||||||
+12
-37
@@ -11,50 +11,25 @@ Thumbs.db
|
|||||||
.vscode/
|
.vscode/
|
||||||
*.sublime-*
|
*.sublime-*
|
||||||
|
|
||||||
# Agent scratch (worktrees of this repo, created and destroyed by
|
|
||||||
# in-flight tooling). Unanchored: .gitignore patterns already match at
|
|
||||||
# every depth, so no prefix is wanted here. This is not a .dockerignore
|
|
||||||
# entry and must not be given a `**/` prefix on the way into one.
|
|
||||||
.claude/
|
|
||||||
|
|
||||||
# Node
|
# Node
|
||||||
node_modules/
|
node_modules/
|
||||||
|
|
||||||
# Secrets. Unanchored like every entry above, so each matches at every
|
# Environment / secrets
|
||||||
# depth. Matching is case-sensitive on Linux, so names use character
|
.env
|
||||||
# ranges rather than a lowercase form that misses `Server.Key`.
|
.env.*
|
||||||
|
*.pem
|
||||||
|
*.key
|
||||||
|
|
||||||
# Environment files. `*.env` covers bare `.env` and the `prod.env`
|
# Go build artifacts
|
||||||
# convention. Only the templates `example.env` and `sample.env` are
|
|
||||||
# re-included below. A repository that commits any other template adds
|
|
||||||
# its own negation at the end of this file, for example `!.env.example`.
|
|
||||||
*.[eE][nN][vV]
|
|
||||||
.[eE][nN][vV].*
|
|
||||||
.[eE][nN][vV][rR][cC]
|
|
||||||
!example.env
|
|
||||||
!sample.env
|
|
||||||
|
|
||||||
# Private keys and the bundles carrying them.
|
|
||||||
*.[pP][eE][mM]
|
|
||||||
*.[kK][eE][yY]
|
|
||||||
*.[pP]12
|
|
||||||
*.[pP][fF][xX]
|
|
||||||
[iI][dD]_[rR][sS][aA]
|
|
||||||
[iI][dD]_[dD][sS][aA]
|
|
||||||
[iI][dD]_[eE][cC][dD][sS][aA]
|
|
||||||
[iI][dD]_[eE][cC][dD][sS][aA]_[sS][kK]
|
|
||||||
[iI][dD]_[eE][dD]25519
|
|
||||||
[iI][dD]_[eE][dD]25519_[sS][kK]
|
|
||||||
|
|
||||||
# This repository's own entries, such as its build outputs, go below
|
|
||||||
# this comment, and a re-vendor keeps them. Anchor a binary built at the
|
|
||||||
# root: `/myapp`, never `myapp`, which also ignores `cmd/myapp/`.
|
|
||||||
/sfdupes
|
/sfdupes
|
||||||
*.log
|
|
||||||
*.out
|
|
||||||
*.test
|
*.test
|
||||||
|
*.out
|
||||||
|
*.log
|
||||||
|
|
||||||
# A scan database lists every path it scanned.
|
# Local scan data
|
||||||
*.sqlite
|
*.sqlite
|
||||||
*.sqlite-shm
|
*.sqlite-shm
|
||||||
*.sqlite-wal
|
*.sqlite-wal
|
||||||
|
|
||||||
|
# Agent worktrees
|
||||||
|
.claude/worktrees/
|
||||||
|
|||||||
+2
-69
@@ -10,23 +10,14 @@ run:
|
|||||||
|
|
||||||
linters:
|
linters:
|
||||||
default: all
|
default: all
|
||||||
enable:
|
|
||||||
# Successor to the deprecated gomodguard. Named explicitly, rather than
|
|
||||||
# left to `default: all`, because it carries the module policy below.
|
|
||||||
- gomodguard_v2
|
|
||||||
disable:
|
disable:
|
||||||
# Genuinely incompatible with project patterns
|
# Genuinely incompatible with project patterns
|
||||||
- exhaustruct # Requires all struct fields
|
- exhaustruct # Requires all struct fields
|
||||||
- exhaustruct_v5 # Requires all struct fields (successor to exhaustruct)
|
- depguard # Dependency allow/block lists
|
||||||
- godot # Requires comments to end with periods
|
- godot # Requires comments to end with periods
|
||||||
|
- wsl # Deprecated, replaced by wsl_v5
|
||||||
- wrapcheck # Too verbose for internal packages
|
- wrapcheck # Too verbose for internal packages
|
||||||
- varnamelen # Short names like db, id are idiomatic Go
|
- varnamelen # Short names like db, id are idiomatic Go
|
||||||
# Deprecated: the warning is attached to the old name, so it is
|
|
||||||
# silenced by disabling that name, not by enabling the successor.
|
|
||||||
- wsl # Deprecated, replaced by wsl_v5
|
|
||||||
- gomodguard # Deprecated, replaced by gomodguard_v2
|
|
||||||
# Misses findings at random in v2.14.0; back once a pinned release fixes it
|
|
||||||
- canonicalheader
|
|
||||||
settings:
|
settings:
|
||||||
lll:
|
lll:
|
||||||
line-length: 88
|
line-length: 88
|
||||||
@@ -37,64 +28,6 @@ linters:
|
|||||||
max-complexity: 15
|
max-complexity: 15
|
||||||
dupl:
|
dupl:
|
||||||
threshold: 100
|
threshold: 100
|
||||||
depguard:
|
|
||||||
# Test-support code must not be compiled into the shipped binary. A
|
|
||||||
# test-support package exists to hand a test privileges the program
|
|
||||||
# itself must never have, so a file that is not a test must not import
|
|
||||||
# one. Test files, and the files inside a package whose directory name
|
|
||||||
# ends in `test`, are where that code belongs, and are exempt.
|
|
||||||
#
|
|
||||||
# The deny list below is the one part of this file a repository is
|
|
||||||
# expected to extend, and the only part it may. depguard matches an
|
|
||||||
# import path against a list of prefixes, so it cannot be told "any path
|
|
||||||
# whose last segment ends in test"; a repository's own test-support
|
|
||||||
# packages have to be named here one at a time, by full import path,
|
|
||||||
# under a module path that differs from repository to repository. Add
|
|
||||||
# them; change nothing else.
|
|
||||||
rules:
|
|
||||||
test-support:
|
|
||||||
list-mode: lax
|
|
||||||
files:
|
|
||||||
- "$all"
|
|
||||||
- "!$test"
|
|
||||||
- "!**/*test/**"
|
|
||||||
deny:
|
|
||||||
- pkg: net/http/httptest
|
|
||||||
desc: >-
|
|
||||||
Test-support code belongs in test files and in packages whose
|
|
||||||
directory name ends in test, not in the shipped binary.
|
|
||||||
# Only decisions already recorded in the Go package defaults are
|
|
||||||
# listed here. Every entry matches the module path exactly.
|
|
||||||
gomodguard_v2:
|
|
||||||
blocked:
|
|
||||||
- module: github.com/rs/zerolog
|
|
||||||
recommendations:
|
|
||||||
- log/slog
|
|
||||||
reason: "Structured logging is stdlib log/slog."
|
|
||||||
# One entry per pre-fork module path, because the later releases
|
|
||||||
# are separate paths. A prefix match would be shorter but would
|
|
||||||
# also reach github.com/go-redis/redismock, the test double for
|
|
||||||
# the successor these entries recommend.
|
|
||||||
- module: github.com/go-redis/redis
|
|
||||||
recommendations:
|
|
||||||
- github.com/redis/go-redis/v9
|
|
||||||
reason: "Pre-fork module; use the maintained go-redis v9."
|
|
||||||
- module: github.com/go-redis/redis/v7
|
|
||||||
recommendations:
|
|
||||||
- github.com/redis/go-redis/v9
|
|
||||||
reason: "Pre-fork module; use the maintained go-redis v9."
|
|
||||||
- module: github.com/go-redis/redis/v8
|
|
||||||
recommendations:
|
|
||||||
- github.com/redis/go-redis/v9
|
|
||||||
reason: "Pre-fork module; use the maintained go-redis v9."
|
|
||||||
- module: github.com/sergi/go-diff
|
|
||||||
recommendations:
|
|
||||||
- github.com/aymanbagabas/go-udiff
|
|
||||||
reason: "No unified diff output; use go-udiff."
|
|
||||||
- module: github.com/hexops/gotextdiff
|
|
||||||
recommendations:
|
|
||||||
- github.com/aymanbagabas/go-udiff
|
|
||||||
reason: "Unmaintained fork; use go-udiff."
|
|
||||||
|
|
||||||
issues:
|
issues:
|
||||||
max-issues-per-linter: 0
|
max-issues-per-linter: 0
|
||||||
|
|||||||
@@ -1,2 +0,0 @@
|
|||||||
node_modules/
|
|
||||||
yarn.lock
|
|
||||||
@@ -1,4 +0,0 @@
|
|||||||
{
|
|
||||||
"tabWidth": 4,
|
|
||||||
"proseWrap": "always"
|
|
||||||
}
|
|
||||||
+99
-58
@@ -1,75 +1,116 @@
|
|||||||
# Lint phase, built alone by script/lint. The tools are invoked directly
|
# Lint stage — fast feedback on formatting and lint issues
|
||||||
# rather than through `make lint`, which runs docker itself and so cannot
|
# golangci/golangci-lint:v2.12.2, 2026-08-07
|
||||||
# run inside a build step.
|
FROM golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240 AS lint
|
||||||
# golangci/golangci-lint:v2.14.0, 2026-10-07
|
|
||||||
FROM golangci/golangci-lint@sha256:ad862ba6b3798cbe0fd9fd7408d498fd74fbd2623a92406b2fd3898faf0bf98f AS lint
|
|
||||||
WORKDIR /src
|
WORKDIR /src
|
||||||
COPY go.mod go.sum ./
|
COPY go.mod go.sum ./
|
||||||
RUN go mod download
|
RUN go mod download
|
||||||
COPY . .
|
COPY . .
|
||||||
|
|
||||||
# The gofmt half of `make fmt-check`. gofmt's output is assigned to a
|
# Cache-buster for the gate layers, and only for them. Docker
|
||||||
# variable first so that its own exit status, as when it cannot parse a
|
# invalidates COPY only when the copied content changes, so on an
|
||||||
# file, still fails the step.
|
# unchanged tree the gates below would be served from cache and the
|
||||||
RUN files="$(gofmt -s -l .)" && \
|
# build would exit 0 having run nothing. script/cibuild and
|
||||||
if [ -n "$files" ]; then \
|
# script/docker pass a fresh CHECK_EPOCH on every invocation.
|
||||||
echo "gofmt: files not formatted:" >&2; echo "$files" >&2; exit 1; \
|
|
||||||
fi
|
|
||||||
|
|
||||||
# Validates .golangci.yml against the schema the pinned binary embeds.
|
|
||||||
RUN golangci-lint config verify --config .golangci.yml
|
|
||||||
RUN golangci-lint run --config .golangci.yml ./...
|
|
||||||
|
|
||||||
# Test phase, built alone by script/test. -race needs cgo and so a C
|
|
||||||
# compiler, which the Debian Go image ships and the alpine one does not.
|
|
||||||
#
|
#
|
||||||
# The tests run as nobody: several of them make a file unreadable and
|
# Two properties this depends on. ARG is per-stage, so the build stage
|
||||||
# expect reading it to fail, and root reads it anyway. nobody has no home
|
# below declares it again; one declaration here would leave that
|
||||||
# directory, so HOME is /tmp, where Go puts its build cache.
|
# stage's gate cacheable. And each gate RUN must reference the value,
|
||||||
# golang:1.25-trixie, 2026-10-04
|
# because BuildKit hashes the expanded command: a declared but
|
||||||
FROM golang@sha256:2c4c60ef415fbfa5e90300722293bef36c5e63fae17570ce18f580af933dbd73 AS test
|
# unreferenced ARG invalidates nothing.
|
||||||
USER nobody
|
#
|
||||||
ENV HOME=/tmp
|
# It sits below the dependency layers deliberately. Everything above it
|
||||||
WORKDIR /src
|
# (the pinned base image, go mod download) keeps its cache; only the
|
||||||
COPY go.mod go.sum ./
|
# gates go cold.
|
||||||
RUN go mod download
|
ARG CHECK_EPOCH
|
||||||
COPY . .
|
|
||||||
RUN go test -timeout 90s -race -cover ./... || \
|
|
||||||
{ echo "--- Rerunning with -v for details ---"; \
|
|
||||||
go test -timeout 90s -race -v ./...; exit 1; }
|
|
||||||
|
|
||||||
# Build stage. Nothing is wanted from either phase above; the copies are
|
# The linter is invoked directly here, not through `make lint`. That
|
||||||
# what make BuildKit build them first, so this stage cannot run unless
|
# target now runs `docker build -f Dockerfile.lint`, and a docker build
|
||||||
# lint and test passed.
|
# cannot run a docker build: routing the gate through make would mean
|
||||||
|
# nesting docker inside this image. Same reason `make check` is gone
|
||||||
|
# from the build stage below. `make fmt-check` stays as it is — it is a
|
||||||
|
# gate, not the aggregate, and it shells out to nothing.
|
||||||
|
RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
|
||||||
|
|
||||||
|
# The FROM above and the one in Dockerfile.lint pin the same linter
|
||||||
|
# twice, and nothing else keeps them in sync; this fails the build when
|
||||||
|
# they disagree. See the script for why it restates neither pin.
|
||||||
|
RUN echo "gate lint-image-pin, epoch ${CHECK_EPOCH}" && \
|
||||||
|
script/verify-lint-image-pin
|
||||||
|
|
||||||
|
# Same config-schema check Dockerfile.lint runs, kept here so this build
|
||||||
|
# gates on exactly what script/lint gates on. It validates against a
|
||||||
|
# schema the pinned binary embeds, so it needs no network.
|
||||||
|
RUN echo "gate config verify, epoch ${CHECK_EPOCH}" && \
|
||||||
|
golangci-lint config verify --config .golangci.yml
|
||||||
|
|
||||||
|
RUN echo "gate lint, epoch ${CHECK_EPOCH}" && \
|
||||||
|
golangci-lint run --config .golangci.yml ./...
|
||||||
|
|
||||||
|
# Build stage
|
||||||
# golang:1.25-alpine, 2026-07-23
|
# golang:1.25-alpine, 2026-07-23
|
||||||
FROM golang@sha256:56961d79ea8129efddcc0b8643fd8a5416b4e6228cfd477e3fd61deb2672c587 AS builder
|
FROM golang@sha256:56961d79ea8129efddcc0b8643fd8a5416b4e6228cfd477e3fd61deb2672c587 AS builder
|
||||||
COPY --from=lint /src/go.sum /dev/null
|
|
||||||
COPY --from=test /src/go.sum /dev/null
|
# We never build or run as root. Create an unprivileged user and point
|
||||||
RUN apk add --no-cache git make
|
# HOME and the Go caches at its home so go build and go test can write
|
||||||
# A tar-stream context keeps the sender's file owners, which git refuses.
|
# their caches when we drop to it below. $GOPATH/bin is deliberately not
|
||||||
RUN git config --system --add safe.directory /src
|
# on PATH: script/bootstrap no longer `go install`s anything (the linter
|
||||||
|
# runs from a pinned image, never from a host install), so nothing lands
|
||||||
|
# there and adding it would only widen what this image resolves.
|
||||||
|
RUN adduser -D -u 1000 builder
|
||||||
|
ENV HOME=/home/builder
|
||||||
|
ENV GOPATH=/home/builder/go
|
||||||
|
ENV GOCACHE=/home/builder/.cache/go-build
|
||||||
|
|
||||||
WORKDIR /src
|
WORKDIR /src
|
||||||
|
|
||||||
|
# No-op file copy whose only purpose is the build-graph edge: it is what
|
||||||
|
# makes this stage depend on the lint stage, and so what forces BuildKit
|
||||||
|
# to finish fmt-check, the pin guard and lint before compilation and
|
||||||
|
# tests start. Remove it and the fail-fast design dies silently — the
|
||||||
|
# build stops gating on lint and still exits 0. It replaces a copy of
|
||||||
|
# the linter binary itself, which is no longer wanted here: nothing in
|
||||||
|
# this stage runs the linter, because `make lint` is now a docker build
|
||||||
|
# and a docker build cannot run inside one.
|
||||||
|
COPY --from=lint /src/go.sum /dev/null
|
||||||
|
|
||||||
|
# Install development prerequisites the same way a developer does,
|
||||||
|
# rather than duplicating the installs inline. Only script/ and the
|
||||||
|
# dependency manifests are copied first, nothing else, so this layer
|
||||||
|
# stays cached until the scripts or the dependencies change — bootstrap
|
||||||
|
# ends in `go mod download`, which is why there is no separate
|
||||||
|
# invocation of it here.
|
||||||
|
COPY script/ script/
|
||||||
COPY go.mod go.sum ./
|
COPY go.mod go.sum ./
|
||||||
RUN go mod download
|
RUN script/bootstrap
|
||||||
|
|
||||||
COPY . .
|
COPY . .
|
||||||
|
|
||||||
# The version stamped into the binary: the VERSION build argument when
|
# Hand the sources and caches to the unprivileged user, then drop root
|
||||||
# one is given, otherwise `git describe --tags --always` of the .git in
|
# before running any checks or builds.
|
||||||
# the build context. A context that carries .git and still yields no
|
RUN chown -R builder:builder /src /home/builder
|
||||||
# version fails the build; with neither, as from a source tarball, it is
|
USER builder
|
||||||
# "dev". `make build` rather than `go build`, so the image and a host
|
|
||||||
# build share one compile recipe, cgo disabled included.
|
|
||||||
ARG VERSION
|
|
||||||
RUN version="${VERSION:-$(git describe --tags --always || echo dev)}"; \
|
|
||||||
if [ -e .git ] && { [ -z "$version" ] || [ "$version" = dev ] || \
|
|
||||||
[ "$version" = unknown ]; }; then \
|
|
||||||
echo "no version could be derived although the build context carries .git" >&2; \
|
|
||||||
exit 1; \
|
|
||||||
fi; \
|
|
||||||
make build VERSION="$version"
|
|
||||||
|
|
||||||
# Runtime stage, and the last one: a plain `docker build .` builds this
|
# Fail the build unless the branch is green. Runs as non-root so the
|
||||||
# stage's chain and nothing else.
|
# permission-denied test paths are exercised legitimately (root would
|
||||||
|
# bypass the chmod(0) the tests rely on).
|
||||||
|
#
|
||||||
|
# The gates are the individual targets, not `make check`: that aggregate
|
||||||
|
# runs `script/lint`, which is now a docker build, and nothing inside an
|
||||||
|
# image build may shell out to docker. Lint is not skipped by this — it
|
||||||
|
# ran in the lint stage above, which this stage's COPY --from makes a
|
||||||
|
# prerequisite. `make`, not the scripts directly, because the Makefile's
|
||||||
|
# `export CGO_ENABLED = 0` applies only to what it invokes.
|
||||||
|
#
|
||||||
|
# Second per-stage declaration of the gate cache-buster; see the lint
|
||||||
|
# stage above for why one is not enough. It is placed after USER so the
|
||||||
|
# drop to the unprivileged user still happens before the checks run.
|
||||||
|
ARG CHECK_EPOCH
|
||||||
|
RUN echo "gate test, epoch ${CHECK_EPOCH}" && make test
|
||||||
|
RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
|
||||||
|
|
||||||
|
RUN make build
|
||||||
|
|
||||||
|
# Runtime stage
|
||||||
# alpine:3.22, 2026-07-23
|
# alpine:3.22, 2026-07-23
|
||||||
FROM alpine@sha256:14358309a308569c32bdc37e2e0e9694be33a9d99e68afb0f5ff33cc1f695dce
|
FROM alpine@sha256:14358309a308569c32bdc37e2e0e9694be33a9d99e68afb0f5ff33cc1f695dce
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,59 @@
|
|||||||
|
# Lint-only image: this is how the linter runs, everywhere. The repo is
|
||||||
|
# COPYed into the pinned golangci-lint image and the linter runs as a
|
||||||
|
# build step, so a successful build IS a clean lint. golangci-lint is
|
||||||
|
# never installed on a host — one toolchain, pinned by digest, identical
|
||||||
|
# on a laptop and in CI — and this works even when the docker daemon is
|
||||||
|
# remote and bind mounts are impossible.
|
||||||
|
#
|
||||||
|
# script/lint builds this file. It is a separate image from the lint
|
||||||
|
# stage of the main Dockerfile because script/lint must not depend on
|
||||||
|
# the rest of that build; the two FROM lines are kept identical by
|
||||||
|
# script/verify-lint-image-pin, run as a gate below.
|
||||||
|
# golangci/golangci-lint:v2.12.2, 2026-08-07
|
||||||
|
FROM golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240
|
||||||
|
|
||||||
|
WORKDIR /src
|
||||||
|
|
||||||
|
# Dependency layers first, so they stay cached across lint runs.
|
||||||
|
COPY go.mod go.sum ./
|
||||||
|
RUN go mod download
|
||||||
|
|
||||||
|
COPY . .
|
||||||
|
|
||||||
|
# Cache-buster for the gate layers, and only for them. Caching of the
|
||||||
|
# lint run is waived by ruling: COPY is invalidated only by changed
|
||||||
|
# content, so on an unchanged tree the gates below would be served from
|
||||||
|
# cache and this build would exit 0 in under a second having run no
|
||||||
|
# linter at all. That exact false green has bitten this repo twice
|
||||||
|
# already (#32, #39). script/lint passes a fresh value on every
|
||||||
|
# invocation.
|
||||||
|
#
|
||||||
|
# Each gate RUN must reference the value, because BuildKit hashes the
|
||||||
|
# expanded command and not the ARG declaration: a declared but
|
||||||
|
# unreferenced ARG invalidates nothing. The ARG sits below the
|
||||||
|
# dependency layers deliberately — everything above it keeps its cache,
|
||||||
|
# only the gates go cold.
|
||||||
|
ARG CHECK_EPOCH
|
||||||
|
|
||||||
|
# The linter version is pinned in two places, here and in the main
|
||||||
|
# Dockerfile's lint stage. Nothing else keeps them in sync, so a
|
||||||
|
# half-applied bump is a build failure; see the script.
|
||||||
|
RUN echo "gate lint-image-pin, epoch ${CHECK_EPOCH}" && \
|
||||||
|
script/verify-lint-image-pin
|
||||||
|
|
||||||
|
# Validates .golangci.yml against golangci-lint's JSON schema. The
|
||||||
|
# concern about this step was that it fetches that schema over a live,
|
||||||
|
# unpinned HTTPS call; measured on the pinned image, it does not. The
|
||||||
|
# binary carries the schema for its own version, so under
|
||||||
|
# `--network none` this both passes on a valid config and still rejects
|
||||||
|
# an invalid one with the jsonschema error. That holds for the gate
|
||||||
|
# steps generally — none of them makes a network call — but not for
|
||||||
|
# this build as a whole: `go mod download` above needs the network on a
|
||||||
|
# cold cache, and under `--network none` a first build fails there
|
||||||
|
# before reaching any gate. That layer stays cached, so only a warm
|
||||||
|
# cache lints offline, until go.mod or go.sum changes.
|
||||||
|
RUN echo "gate config verify, epoch ${CHECK_EPOCH}" && \
|
||||||
|
golangci-lint config verify --config .golangci.yml
|
||||||
|
|
||||||
|
RUN echo "gate lint, epoch ${CHECK_EPOCH}" && \
|
||||||
|
golangci-lint run --config .golangci.yml ./...
|
||||||
+82
-386
@@ -1,6 +1,6 @@
|
|||||||
---
|
---
|
||||||
title: Repository Policies
|
title: Repository Policies
|
||||||
last_modified: 2026-10-07
|
last_modified: 2026-07-06
|
||||||
---
|
---
|
||||||
|
|
||||||
This document covers repository structure, tooling, and workflow standards. Code
|
This document covers repository structure, tooling, and workflow standards. Code
|
||||||
@@ -60,28 +60,17 @@ style conventions are in separate documents:
|
|||||||
prerequisite since nvm requires bash. yarn is then pinned via
|
prerequisite since nvm requires bash. yarn is then pinned via
|
||||||
`corepack prepare yarn@<version> --activate`. Never install "latest" or "lts";
|
`corepack prepare yarn@<version> --activate`. Never install "latest" or "lts";
|
||||||
always exact versions. `script/cibuild` runs the CI build: it changes to the
|
always exact versions. `script/cibuild` runs the CI build: it changes to the
|
||||||
repo root, runs `script/bootstrap`, runs `script/check`, and builds the image
|
repo root and runs `docker build .`; the Gitea workflow calls it. Four further
|
||||||
with the version; the Gitea workflow calls it. **`script/cibuild` runs
|
scripts are our own extensions to the standard: `script/check` runs
|
||||||
`script/bootstrap` first**, because the workflow checks out the repo and runs
|
`script/test`, `script/lint`, and `script/fmt-check`; `script/precommit` is
|
||||||
nothing else, while `script/fmt-check` runs the formatter on the host: on a
|
what the git pre-commit hook runs, and it calls `script/check`;
|
||||||
pristine checkout with nothing installed the run dies there, after the
|
`script/install-precommit` installs the git pre-commit hook (the `make hooks`
|
||||||
containerised gates have passed. **The bootstrap alone is not enough**:
|
target shims to it); and `script/projectname` (literally that filename) simply
|
||||||
`script/bootstrap` installs node and yarn under nvm and leaves neither on the
|
outputs the project's name. Scripts that need the name call
|
||||||
`PATH` of the shell that called it, so a bare `yarn` still exits 127. The host
|
`script/projectname` — e.g. `script/docker` assembles its image tag from it —
|
||||||
entrypoints that need yarn — `script/fmt` and `script/fmt-check` — therefore
|
so those scripts stay byte-identical across all repos. Repo-type-specific
|
||||||
source nvm for the pinned node version before invoking it, exactly as
|
pre-commit extras (e.g. `go mod tidy` verification in Go repos) belong in
|
||||||
`script/bootstrap`'s own install step does. A runner carrying nothing but
|
`script/precommit`, not in the hook itself. Model scripts are at
|
||||||
docker and git then gets through `script/check`. Four further scripts are our
|
|
||||||
own extensions to the standard: `script/check` runs `script/test`,
|
|
||||||
`script/lint` and `script/fmt-check`; `script/precommit` is what the git
|
|
||||||
pre-commit hook runs, and it calls `script/check`; `script/install-precommit`
|
|
||||||
installs the git pre-commit hook (the `make hooks` target shims to it); and
|
|
||||||
`script/projectname` (literally that filename) simply outputs the project's
|
|
||||||
name. Scripts that need the name call `script/projectname` — e.g.
|
|
||||||
`script/docker` assembles its image tag from it — so those scripts stay
|
|
||||||
byte-identical across all repos. Repo-type-specific pre-commit extras (e.g.
|
|
||||||
`go mod tidy` verification in Go repos) belong in `script/precommit`, not in
|
|
||||||
the hook itself. Model scripts are at
|
|
||||||
`https://git.eeqj.de/sneak/prompts/raw/branch/main/script/<name>`. The README
|
`https://git.eeqj.de/sneak/prompts/raw/branch/main/script/<name>`. The README
|
||||||
must document the provided scripts in an **Entrypoints** section (see the
|
must document the provided scripts in an **Entrypoints** section (see the
|
||||||
README requirements below).
|
README requirements below).
|
||||||
@@ -100,222 +89,87 @@ style conventions are in separate documents:
|
|||||||
contributor should be able to understand the entire development workflow by
|
contributor should be able to understand the entire development workflow by
|
||||||
reading the Makefile.
|
reading the Makefile.
|
||||||
|
|
||||||
- Every repo should have a `Dockerfile`, and it carries the repo's gates: a
|
- Every repo should have a `Dockerfile`. All Dockerfiles must run `make check`
|
||||||
`lint` phase and a `test` phase, with the final stage depending on both so the
|
as a build step so the build fails if the branch is not green. For non-server
|
||||||
image cannot be built unless they pass. For non-server repos the final stage
|
repos, the Dockerfile should bring up a development environment and run
|
||||||
brings up a development environment; for server repos it is the runtime image.
|
`make check`. For server repos, `make check` should run as an early build
|
||||||
The gate phases and the build stage start from their pinned base images and
|
stage before the final image is assembled. Dockerfiles install development
|
||||||
install what those images lack either inline, as the canonical Go `Dockerfile`
|
prerequisites by running `script/bootstrap` rather than duplicating installs
|
||||||
below does for `git`, or by running `script/bootstrap`, as the `prompts`
|
inline; COPY `script/` and the dependency manifests (`package.json` +
|
||||||
repo's own `Dockerfile` does for its yarn packages. The development
|
`yarn.lock`, `go.mod` + `go.sum`, etc.) before running it so the bootstrap
|
||||||
environment stage installs development prerequisites by running
|
layer stays cached until dependencies change.
|
||||||
`script/bootstrap` rather than duplicating its installs inline. A stage that
|
|
||||||
runs `script/bootstrap` COPYs `script/` and the dependency manifests
|
|
||||||
(`package.json` + `yarn.lock`, `go.mod` + `go.sum`, etc.) before running it.
|
|
||||||
|
|
||||||
- **Linting and testing run in Docker, as phases of the `Dockerfile`.** There is
|
- **Dockerfiles must use a separate lint stage for fail-fast feedback.** Go
|
||||||
no separate lint file. `script/lint` and `script/test` each build one phase
|
repos use a multistage build where linting runs in an independent stage based
|
||||||
and nothing else:
|
on the `golangci/golangci-lint` image (pinned by hash). This stage runs
|
||||||
|
`make fmt-check` and `make lint` before the full build begins. The build stage
|
||||||
|
then declares an explicit dependency on the lint stage via
|
||||||
|
`COPY --from=lint /src/go.sum /dev/null`, which forces BuildKit to complete
|
||||||
|
linting before proceeding to compilation and tests. This ensures lint failures
|
||||||
|
surface in seconds rather than minutes, without blocking on dependency
|
||||||
|
download or compilation in the build stage.
|
||||||
|
|
||||||
```sh
|
The standard pattern for a Go repo Dockerfile is:
|
||||||
docker build --no-cache --target lint --output type=cacheonly .
|
|
||||||
docker build --no-cache --target test --output type=cacheonly .
|
|
||||||
```
|
|
||||||
|
|
||||||
**A stage that is not the last one in the file is built only when the final
|
|
||||||
stage's chain depends on it, or when `--target` names it.** That is why the
|
|
||||||
two gates are always invoked by name here, and why the final stage carries a
|
|
||||||
`COPY --from=` of a harmless file from each of them: without that edge a
|
|
||||||
plain `docker build .` builds the last stage alone and exits 0 having linted
|
|
||||||
and tested nothing.
|
|
||||||
|
|
||||||
**The gate builds write no image.** With `--output type=cacheonly` the phase
|
|
||||||
runs and a failing step fails the build, but the result is not exported.
|
|
||||||
Nothing uses those images, and writing one out is slow: a Go test phase's
|
|
||||||
image holds the toolchain and every compiled package. A build given neither
|
|
||||||
`--output` nor `-t` writes an untagged image and leaves it dangling, on
|
|
||||||
every developer host and every CI runner. `script/cibuild` and
|
|
||||||
`script/docker` build the image that ships and tag it, so each build
|
|
||||||
replaces the previous image; each assigns the tag on its own line before the
|
|
||||||
build, so `set -e` stops it where `script/projectname` fails.
|
|
||||||
|
|
||||||
Inside a phase the tool is invoked directly — `golangci-lint`, `go test`,
|
|
||||||
`eslint`, `prettier` — never through `make lint` or `script/test`, which are
|
|
||||||
themselves a `docker build` and would recurse into a daemon that does not
|
|
||||||
exist in a build step. Formatting is the exception and stays on the host:
|
|
||||||
`script/fmt` writes the working tree, and `script/fmt-check` is its
|
|
||||||
read-only twin.
|
|
||||||
|
|
||||||
**No lint verdict may come from a host invocation of the linter.** On a
|
|
||||||
shared host golangci-lint reads a result cache keyed on file content rather
|
|
||||||
than location, so a second checkout of the same content is served the first
|
|
||||||
one's findings, and a host-global lock in `$TMPDIR` makes concurrent runs
|
|
||||||
exit non-zero with `parallel golangci-lint is running` — a status a caller
|
|
||||||
cannot tell from real findings. Both have produced wrong verdicts in this
|
|
||||||
org, in both directions. A container has its own cache, its own `TMPDIR` and
|
|
||||||
a digest-pinned binary, so neither is reachable.
|
|
||||||
|
|
||||||
- **Any build that runs checks is built with `--no-cache`.** Docker invalidates
|
|
||||||
a `COPY` layer only when the copied content changes, so on an unchanged tree
|
|
||||||
the check `RUN` is served from cache, nothing executes, and the build still
|
|
||||||
exits 0. Every `docker build` in `script/` therefore passes `--no-cache`:
|
|
||||||
`script/lint`, `script/test`, `script/cibuild` and `script/docker` are the
|
|
||||||
four, and there is no fifth — `script/check` runs the two gate phases and
|
|
||||||
`script/fmt-check`, and builds no image of its own. A bare `docker build .` is
|
|
||||||
not evidence that anything ran: a sub-second build reporting success is a
|
|
||||||
cache hit, not a result. Never invalidate by pruning — `docker builder prune`
|
|
||||||
and friends destroy a build cache shared with every other build on the host.
|
|
||||||
When a check is added or changed, prove it works by planting a defect it must
|
|
||||||
catch and watching the run fail on it, then revert the defect. A green run
|
|
||||||
alone shows neither that the check ran nor that it covers what it should.
|
|
||||||
|
|
||||||
- **The gate phases are separate stages, and the build stage depends on both.**
|
|
||||||
The lint phase is based on the `golangci/golangci-lint` image (pinned by
|
|
||||||
hash), so lint failures surface in seconds rather than after a full compile,
|
|
||||||
and the test phase is based on the Debian Go image. The canonical Go repo
|
|
||||||
`Dockerfile`:
|
|
||||||
|
|
||||||
```dockerfile
|
```dockerfile
|
||||||
# Lint phase
|
# Lint stage — fast feedback on formatting and lint issues
|
||||||
# golangci/golangci-lint:v2.x.x, YYYY-MM-DD
|
# golangci/golangci-lint:v2.x.x, YYYY-MM-DD
|
||||||
FROM golangci/golangci-lint@sha256:... AS lint
|
FROM golangci/golangci-lint@sha256:... AS lint
|
||||||
WORKDIR /src
|
WORKDIR /src
|
||||||
COPY go.mod go.sum ./
|
COPY go.mod go.sum ./
|
||||||
RUN go mod download
|
RUN go mod download
|
||||||
COPY . .
|
COPY . .
|
||||||
RUN golangci-lint run --config .golangci.yml ./...
|
RUN make fmt-check
|
||||||
|
RUN make lint
|
||||||
|
|
||||||
# Test phase. -race needs cgo and so a C compiler, which the Debian Go
|
# Build stage
|
||||||
# image ships and the alpine one does not.
|
|
||||||
# golang:1.x, YYYY-MM-DD
|
|
||||||
FROM golang@sha256:... AS test
|
|
||||||
WORKDIR /src
|
|
||||||
COPY go.mod go.sum ./
|
|
||||||
RUN go mod download
|
|
||||||
COPY . .
|
|
||||||
RUN go test -timeout 90s -race -cover ./... || \
|
|
||||||
{ echo "--- Rerunning with -v for details ---"; \
|
|
||||||
go test -timeout 90s -race -v ./...; exit 1; }
|
|
||||||
|
|
||||||
# Build stage. Nothing is wanted from either phase above; the copies
|
|
||||||
# are what make BuildKit build them first, so this stage cannot run
|
|
||||||
# unless lint and test passed.
|
|
||||||
# golang:1.x-alpine, YYYY-MM-DD
|
# golang:1.x-alpine, YYYY-MM-DD
|
||||||
FROM golang@sha256:... AS builder
|
FROM golang@sha256:... AS builder
|
||||||
COPY --from=lint /src/go.sum /dev/null
|
|
||||||
COPY --from=test /src/go.sum /dev/null
|
|
||||||
RUN apk add --no-cache git
|
|
||||||
# A tar-stream context keeps the sender's file owners, which git refuses.
|
|
||||||
RUN git config --system --add safe.directory /src
|
|
||||||
WORKDIR /src
|
WORKDIR /src
|
||||||
|
|
||||||
|
# Force BuildKit to run the lint stage before proceeding
|
||||||
|
COPY --from=lint /src/go.sum /dev/null
|
||||||
|
|
||||||
COPY go.mod go.sum ./
|
COPY go.mod go.sum ./
|
||||||
RUN go mod download
|
RUN go mod download
|
||||||
COPY . .
|
COPY . .
|
||||||
|
RUN make test
|
||||||
|
|
||||||
# The VERSION build arg when one is given, otherwise
|
ARG VERSION=dev
|
||||||
# `git describe --tags --always` on the .git in the build context. With
|
RUN CGO_ENABLED=0 go build -trimpath \
|
||||||
# .git present, a version that is still empty, dev or unknown fails the
|
|
||||||
# build: git is missing or could not read the checkout.
|
|
||||||
ARG VERSION
|
|
||||||
RUN VERSION="${VERSION:-$(git describe --tags --always)}"; \
|
|
||||||
if [ -e .git ]; then \
|
|
||||||
case "$VERSION" in ""|dev|unknown) \
|
|
||||||
echo "version is '$VERSION' although .git is present" >&2; \
|
|
||||||
exit 1 ;; \
|
|
||||||
esac; \
|
|
||||||
fi; \
|
|
||||||
CGO_ENABLED=0 go build -trimpath \
|
|
||||||
-ldflags="-s -w -X main.Version=${VERSION}" \
|
-ldflags="-s -w -X main.Version=${VERSION}" \
|
||||||
-o /app ./cmd/app/
|
-o /app ./cmd/app/
|
||||||
|
|
||||||
# Runtime stage, and the last one
|
# Runtime stage
|
||||||
FROM alpine@sha256:...
|
FROM alpine@sha256:...
|
||||||
COPY --from=builder /app /usr/local/bin/app
|
COPY --from=builder /app /usr/local/bin/app
|
||||||
ENTRYPOINT ["app"]
|
ENTRYPOINT ["app"]
|
||||||
```
|
```
|
||||||
|
|
||||||
Key points:
|
Key points:
|
||||||
- The lint phase uses the `golangci/golangci-lint` image directly (it has
|
- The lint stage uses the `golangci/golangci-lint` image directly (it
|
||||||
both Go and the linter), so nothing needs installing.
|
includes both Go and the linter), so there is no need to install the
|
||||||
- `COPY --from=<phase> /src/go.sum /dev/null` is a no-op copy whose only
|
linter separately.
|
||||||
purpose is the ordering edge. BuildKit runs stages in parallel by default,
|
- `COPY --from=lint /src/go.sum /dev/null` is a no-op file copy that creates
|
||||||
and a stage nothing depends on is not built at all, so without these two
|
a stage dependency. BuildKit runs stages in parallel by default; without
|
||||||
lines a red gate would not fail the build.
|
this line, the build stage would not wait for lint to finish and a lint
|
||||||
- Keep the runtime stage last, and if you add a stage after it, give it the
|
failure might not fail the overall build.
|
||||||
same two copies. A plain `docker build .` builds the last stage's chain
|
|
||||||
and nothing else.
|
|
||||||
- If the project uses `//go:embed` directives that reference build artifacts
|
- If the project uses `//go:embed` directives that reference build artifacts
|
||||||
(e.g. a web frontend compiled in a separate stage), the lint phase must
|
(e.g. a web frontend compiled in a separate stage), the lint stage must
|
||||||
create placeholder files so the embed directives resolve. Example:
|
create placeholder files so the embed directives resolve. Example:
|
||||||
`RUN mkdir -p web/dist && touch web/dist/index.html web/dist/style.css`.
|
`RUN mkdir -p web/dist && touch web/dist/index.html web/dist/style.css`.
|
||||||
- If the project requires CGO or system libraries for linting, install them
|
The lint stage should not depend on the actual build output — it exists to
|
||||||
in the lint phase. The `golangci/golangci-lint` image is Debian-based and
|
fail fast.
|
||||||
has no `apk`, so install with `apt-get` under the Debian package name
|
- If the project requires CGO or system libraries for linting (e.g.
|
||||||
(`libvips-dev`, where alpine says `vips-dev`), and delete the package
|
`vips-dev`), install them in the lint stage with `apk add`.
|
||||||
lists in the same `RUN`, so the layer does not keep them:
|
- The build stage runs `make test` after compilation setup. Tests run in the
|
||||||
|
build stage, not the lint stage, because they may require compiled
|
||||||
```dockerfile
|
artifacts or heavier dependencies.
|
||||||
RUN apt-get update \
|
|
||||||
&& apt-get install -y --no-install-recommends libvips-dev \
|
|
||||||
&& rm -rf /var/lib/apt/lists/*
|
|
||||||
```
|
|
||||||
|
|
||||||
- `.dockerignore` lets `.git` into the build context. It keeps out every git
|
|
||||||
`config` at any depth (`**/.git/config`, `**/.git/modules/**/config`): the
|
|
||||||
repository's own, each submodule's under `.git/modules/`, and that of a
|
|
||||||
submodule keeping its own `.git` directory. `git describe` does not need
|
|
||||||
them, and each can hold a credential: a password in a remote URL, or the
|
|
||||||
token the CI checkout step stores there. A submodule whose name has a
|
|
||||||
`config` segment (`config`, `deploy/config`, `config/lib`) loses its whole
|
|
||||||
git directory to `**/.git/modules/**/config`, and Go's version stamping
|
|
||||||
then fails the build: give it a name without that segment
|
|
||||||
(`git submodule add --name`). The stage that compiles has `git` (the
|
|
||||||
Debian Go image has it; an alpine one needs `apk add --no-cache git`) and
|
|
||||||
takes the version from the `VERSION` build argument when one is given,
|
|
||||||
otherwise from `git describe --tags --always`. That gives the tag on a
|
|
||||||
tagged commit; on a later commit, the tag, the number of commits since it
|
|
||||||
and the short commit (`v1.2.3-4-gabc1234`); and the short commit when no
|
|
||||||
tag is reachable. The stage that compiles also marks its working directory
|
|
||||||
safe for git (`git config --system --add safe.directory /src`): a context
|
|
||||||
sent as a tar stream keeps the sender's file owners, and git refuses a
|
|
||||||
checkout owned by another user, so the version would come out empty.
|
|
||||||
`ARG VERSION` has no default, and the build fails if the context carries
|
|
||||||
`.git` and the version still comes out empty, `dev` or `unknown`. A plain
|
|
||||||
`docker build .` with no build arguments must succeed; a Dockerfile that
|
|
||||||
refuses an empty build argument drops that refusal and keeps the argument.
|
|
||||||
A checkout whose `.git` is a file (a linked worktree, or a repository
|
|
||||||
checked out as a submodule) is the exception: that file points to a git
|
|
||||||
directory outside the build context, so the build cannot read the version
|
|
||||||
and a plain `docker build .` fails; pass the version with
|
|
||||||
`--build-arg VERSION=...`, as `script/docker` and `script/cibuild` already
|
|
||||||
do.
|
|
||||||
|
|
||||||
- Every repo should have a Gitea Actions workflow (`.gitea/workflows/`) that
|
- Every repo should have a Gitea Actions workflow (`.gitea/workflows/`) that
|
||||||
runs `script/cibuild` on push, and checks out the repo as its only other step,
|
runs `script/cibuild` (which runs `docker build .`) on push. Since the
|
||||||
with `persist-credentials: false`: `script/cibuild` needs no token, and
|
Dockerfile already runs `make check`, a successful build implies all checks
|
||||||
without it the checkout leaves the job's token in `.git/config` for every
|
pass.
|
||||||
later step. The checkout step also sets `fetch-depth: 0`, which fetches the
|
|
||||||
tags `git describe` needs: by default it clones shallow with no tags, and a
|
|
||||||
tagged repository's CI build would stamp a bare short commit id. The
|
|
||||||
workflow's `concurrency` block groups runs by workflow and branch
|
|
||||||
(`${{ github.workflow }}-${{ github.ref }}`) with `cancel-in-progress: true`,
|
|
||||||
so a new push cancels the older run on the same branch, queued or running, and
|
|
||||||
no other: runs for replaced commits do not hold up the shared runner.
|
|
||||||
`script/cibuild` bootstraps, runs the gate phases, and then builds the image,
|
|
||||||
so a successful run means every check passed; a bare `docker build .` does not
|
|
||||||
carry the same guarantee, because its gate phases may come from the cache. The
|
|
||||||
image build is uncached and so runs the gate phases a second time. That is the
|
|
||||||
price of the rule above, and it is worth paying: the image that ships is built
|
|
||||||
from a run of its own gates rather than from a cache entry. The `check` job
|
|
||||||
sets `timeout-minutes: 20`, so a hung build frees the shared runner after 20
|
|
||||||
minutes. That allows for the three Docker builds described above (the test
|
|
||||||
phase, the lint phase, then the image), each held to the 5-minute Docker build
|
|
||||||
limit below, plus the bootstrap. A separate workflow limited to `main` by a
|
|
||||||
`branches` list under `on: push` cannot be checked by review: to try a change
|
|
||||||
to it, add the feature branch to that list and push, then remove the branch
|
|
||||||
from the list again before merging. Keep any job in it that publishes behind
|
|
||||||
`if: github.ref_name == 'main'`, so the run from the feature branch publishes
|
|
||||||
nothing.
|
|
||||||
|
|
||||||
- Use platform-standard formatters: `black` for Python, `prettier` for
|
- Use platform-standard formatters: `black` for Python, `prettier` for
|
||||||
JS/CSS/Markdown/HTML, `go fmt` for Go. Always use default configuration with
|
JS/CSS/Markdown/HTML, `go fmt` for Go. Always use default configuration with
|
||||||
@@ -335,21 +189,14 @@ style conventions are in separate documents:
|
|||||||
module under test to verify it compiles/parses. There is no excuse for
|
module under test to verify it compiles/parses. There is no excuse for
|
||||||
`make test` to be a no-op.
|
`make test` to be a no-op.
|
||||||
|
|
||||||
- `make test` must complete in under 60 seconds. That is the hard cap, and a
|
- `make test` must complete in under 20 seconds. Add a 30-second timeout in the
|
||||||
suite that exceeds it fails. Under 20 seconds is the target. A suite between
|
Makefile.
|
||||||
20 and 60 seconds is still green, but the overage must be filed as an
|
|
||||||
improvement bug against that repo. Add a 90-second timeout to the test
|
|
||||||
invocation (`go test -timeout 90s`). The backstop deliberately sits above the
|
|
||||||
hard cap so that it catches a genuinely hung test rather than a merely slow
|
|
||||||
one.
|
|
||||||
|
|
||||||
- **The test command should use the conditional verbose rerun pattern.** Run
|
- **`make test` should use the conditional verbose rerun pattern.** Run tests
|
||||||
tests without `-v` (verbose) first. If tests fail, automatically rerun with
|
without `-v` (verbose) first. If tests fail, automatically rerun with `-v` to
|
||||||
`-v` to show full output. This keeps CI logs and `docker build` output clean
|
show full output. This keeps CI logs and `docker build` output clean on
|
||||||
on success (just package/suite summaries) while providing full diagnostic
|
success (just package/suite summaries) while providing full diagnostic detail
|
||||||
detail on failure (every test case, every assertion). The command lives in the
|
on failure (every test case, every assertion). The general shell pattern:
|
||||||
`test` phase of the `Dockerfile`, since `script/test` builds that phase; the
|
|
||||||
Makefile form below is the same pattern for any repo-local invocation:
|
|
||||||
|
|
||||||
```makefile
|
```makefile
|
||||||
test:
|
test:
|
||||||
@@ -362,26 +209,11 @@ style conventions are in separate documents:
|
|||||||
|
|
||||||
```makefile
|
```makefile
|
||||||
test:
|
test:
|
||||||
@go test -count=1 -timeout 90s -race -cover ./... || \
|
@go test -timeout 30s -race -cover ./... || \
|
||||||
{ echo "--- Rerunning with -v for details ---"; \
|
{ echo "--- Rerunning with -v for details ---"; \
|
||||||
go test -count=1 -timeout 90s -race -v ./...; exit 1; }
|
go test -timeout 30s -race -v ./...; exit 1; }
|
||||||
```
|
```
|
||||||
|
|
||||||
`-count=1` is required on both invocations: it defeats Go's test _result_
|
|
||||||
cache, so neither run can report a stored pass in place of running the
|
|
||||||
tests. It leaves the build cache alone, so it costs the runtime of the suite
|
|
||||||
and no recompilation.
|
|
||||||
|
|
||||||
That cache is Go's own, separate from Docker's layer cache. Go stores a
|
|
||||||
passing result in its cache directory (`GOCACHE`), and when the same tests
|
|
||||||
run again on unchanged code it prints that result, marked `(cached)`,
|
|
||||||
without running them. That matters on a developer's machine, where this
|
|
||||||
target runs and the directory lasts from one run to the next. The `test`
|
|
||||||
phase of the `Dockerfile` needs no `-count=1`: its base image holds no
|
|
||||||
result for this repo's tests and nothing before its `go test` step runs a
|
|
||||||
test, so there is nothing to replay. `--no-cache` (above) is what makes that
|
|
||||||
step run on an unchanged tree.
|
|
||||||
|
|
||||||
Python example:
|
Python example:
|
||||||
|
|
||||||
```makefile
|
```makefile
|
||||||
@@ -407,89 +239,10 @@ style conventions are in separate documents:
|
|||||||
must be in `.gitignore`. No exceptions.
|
must be in `.gitignore`. No exceptions.
|
||||||
|
|
||||||
- `.gitignore` should be comprehensive from the start: OS files (`.DS_Store`),
|
- `.gitignore` should be comprehensive from the start: OS files (`.DS_Store`),
|
||||||
editor files (`.swp`, `*~`), in-repo agent scratch directories (`.claude/`),
|
editor files (`.swp`, `*~`), language build artifacts, and `node_modules/`.
|
||||||
`node_modules/`, and the repo's own build outputs. Fetch the standard
|
Fetch the standard `.gitignore` from
|
||||||
`.gitignore` from
|
|
||||||
`https://git.eeqj.de/sneak/prompts/raw/branch/main/.gitignore` when setting up
|
`https://git.eeqj.de/sneak/prompts/raw/branch/main/.gitignore` when setting up
|
||||||
a new repo. A repo's `.gitignore` is the standard file followed by the repo's
|
a new repo.
|
||||||
own entries, such as its binaries; a re-vendor replaces the standard part and
|
|
||||||
keeps those entries. These patterns are written to `.gitignore`'s own
|
|
||||||
semantics, in which an unanchored pattern already matches at every depth; they
|
|
||||||
are not a `.dockerignore` and must not be transplanted into one unmodified.
|
|
||||||
|
|
||||||
- **`.dockerignore` does not use `.gitignore` semantics, and copying patterns
|
|
||||||
across unmodified leaves secrets in the build context.** Docker matches with
|
|
||||||
`moby/patternmatcher`: `filepath.Match` semantics plus a `**` extension, so
|
|
||||||
`*` does not cross `/` and a pattern without a leading `**/` is anchored at
|
|
||||||
the build-context root. A `.dockerignore` listing `.env`, `*.pem` and `*.key`
|
|
||||||
therefore excludes only the copies at the repository root, while `config/.env`
|
|
||||||
and `certs/server.key` still reach the context and can land in an image layer
|
|
||||||
— which is more dangerous than a short file with no secret patterns at all,
|
|
||||||
because it reads as solved and stops anyone looking. Give every
|
|
||||||
depth-independent pattern the `**/` prefix and leave only genuinely
|
|
||||||
root-anchored entries unprefixed: `.claude`, and the repo's own host-built
|
|
||||||
binary, written `/myapp` and never `**/myapp`, which would also match
|
|
||||||
`cmd/myapp/` and delete the package directory from the context. Matching is
|
|
||||||
case-sensitive, and an ALL-CAPS twin per pattern still misses `Server.Key`, so
|
|
||||||
secret names use character ranges — `**/*.[kK][eE][yY]`, `**/*.[pP][eE][mM]`,
|
|
||||||
and likewise for `.envrc` and the extensionless SSH keys. Where such a pattern
|
|
||||||
also catches something the build needs, re-include it with a negation
|
|
||||||
(`!docs/example.env`); deleting the pattern reopens the exposure for every
|
|
||||||
other file it covers. Fetch the standard `.dockerignore` from
|
|
||||||
`https://git.eeqj.de/sneak/prompts/raw/branch/main/.dockerignore` and extend
|
|
||||||
it with the repo's own artifacts.
|
|
||||||
|
|
||||||
- **In-repo agent scratch belongs in both files, written to each file's own
|
|
||||||
semantics.** `.claude/` holds one worktree per in-flight agent — an entire
|
|
||||||
additional checkout of the repo — so under `COPY . .` the build context
|
|
||||||
inflates by a multiple of the repo and another session's unreviewed work can
|
|
||||||
be copied into an image layer. In `.gitignore` the entry is `.claude/`,
|
|
||||||
unanchored. In `.dockerignore` it is `.claude`, anchored and with **no** `**/`
|
|
||||||
prefix, because the prefixed form would also delete any nested directory of
|
|
||||||
that name from the build. Anchoring carries a known gap that the canonical
|
|
||||||
`.dockerignore` states in its own comment, since consuming repos receive the
|
|
||||||
file and not the tracker: the directory is created in the agent's working
|
|
||||||
directory, so a repo running agents in subdirectories still ships
|
|
||||||
`services/api/.claude/` and must add its own anchored entry there.
|
|
||||||
|
|
||||||
- **A plain `docker build .` of a clone stamps the version that
|
|
||||||
`git describe --tags --always` gives**, derived from the `.git` in the build
|
|
||||||
context as the canonical `Dockerfile` above shows. Without its failure check,
|
|
||||||
a missing `git` or an unreadable checkout would leave `-X main.Version=` empty
|
|
||||||
and the build would still exit 0. `script/docker` and `script/cibuild` pass
|
|
||||||
the version they compute on the host; it takes precedence. They do this
|
|
||||||
byte-identically across repos:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
# The version and the tag each get their own line: a failing command
|
|
||||||
# substitution inside an argument does not trip `set -e`, so the inline
|
|
||||||
# form degrades to an empty constant.
|
|
||||||
version="$(git describe --tags --always --dirty 2>/dev/null || true)"
|
|
||||||
[ -n "$version" ] || version="unknown"
|
|
||||||
tag="$(script/projectname)"
|
|
||||||
docker build --no-cache \
|
|
||||||
--build-arg VERSION="$version" \
|
|
||||||
-t "$tag" .
|
|
||||||
```
|
|
||||||
|
|
||||||
`--always` makes an untagged repo yield an abbreviated commit hash rather
|
|
||||||
than failing, and the `[ -n "$version" ]` line is the single place the
|
|
||||||
fallback is applied — a live check that fires on a build from an export with
|
|
||||||
no `.git` and on a repository with no commits yet. Do not fold it into the
|
|
||||||
substitution as `|| echo unknown`, which makes the guard unreachable. The
|
|
||||||
Dockerfile's side is `ARG VERSION` in the stage that compiles, declared
|
|
||||||
there because `ARG` is stage-scoped; passing `VERSION` to a repo whose
|
|
||||||
Dockerfile declares no such `ARG` is ignored and costs nothing, which is why
|
|
||||||
the scripts stay byte-identical. One consequence for CI: the standard
|
|
||||||
checkout action clones shallow and fetches no tags, so the canonical
|
|
||||||
`.gitea/workflows/check.yml` sets `fetch-depth: 0` on its checkout step.
|
|
||||||
|
|
||||||
- **Verify `.dockerignore` by enumerating the image, not by reading the
|
|
||||||
patterns.** Plant files at the root _and_ at least two directories deep, build
|
|
||||||
a probe image that does `COPY . .`, and list what actually landed
|
|
||||||
(`docker run --rm --entrypoint find IMAGE /app`). The `transferring context`
|
|
||||||
size is not a substitute: a nested secret is a few bytes, and BuildKit
|
|
||||||
transfers only the delta from the previous build.
|
|
||||||
|
|
||||||
- **No build artifacts in version control.** Code-derived data (compiled
|
- **No build artifacts in version control.** Code-derived data (compiled
|
||||||
bundles, minified output, generated assets) must never be committed to the
|
bundles, minified output, generated assets) must never be committed to the
|
||||||
@@ -505,56 +258,9 @@ style conventions are in separate documents:
|
|||||||
- Make all changes on a feature branch. You can do whatever you want on a
|
- Make all changes on a feature branch. You can do whatever you want on a
|
||||||
feature branch.
|
feature branch.
|
||||||
|
|
||||||
- `.golangci.yml` is standardized. The vendored copy in a consuming repo must
|
- `.golangci.yml` is standardized and must _NEVER_ be modified by an agent, only
|
||||||
_NEVER_ be modified by an agent: fetch it from
|
manually by the user. Fetch from
|
||||||
`https://git.eeqj.de/sneak/prompts/raw/branch/main/.golangci.yml` and keep it
|
`https://git.eeqj.de/sneak/prompts/raw/branch/main/.golangci.yml`.
|
||||||
byte-identical, so that no repo can quietly loosen its own linting. Linter
|
|
||||||
configuration changes are made to the canonical copy in the `prompts` repo and
|
|
||||||
reach consuming repos by re-vendoring; an agent may open a PR against
|
|
||||||
canonical, which only the user merges. One list is exempt from byte-identity,
|
|
||||||
because it cannot be written once for every repo: the `deny` list of the
|
|
||||||
`test-support` depguard rule, where a repo names its own test-support packages
|
|
||||||
by full import path. A repo adds entries there and changes nothing else, and a
|
|
||||||
re-vendor carries its entries forward. The canonical golangci-lint version is
|
|
||||||
v2.14.0 (released 2026-09-24), pinned as the digest of the lint phase's base
|
|
||||||
image
|
|
||||||
(`golangci/golangci-lint@sha256:ad862ba6b3798cbe0fd9fd7408d498fd74fbd2623a92406b2fd3898faf0bf98f`,
|
|
||||||
which reports `2.14.0 built with go1.27.0 from 114493f9`). A module's `go`
|
|
||||||
directive must not name a newer Go minor version than the one golangci-lint
|
|
||||||
was built with, or golangci-lint refuses to lint it: this release lints
|
|
||||||
`go 1.27.1` but not `go 1.28`. That digest is the only pin, since no repo
|
|
||||||
installs golangci-lint on the host. A repo sets the lint phase digest to the
|
|
||||||
one named here and re-vendors `.golangci.yml` in the same commit, whichever of
|
|
||||||
the two prompted the change: the canonical copy can name linters that an older
|
|
||||||
golangci-lint rejects, and a newer golangci-lint can add linters that
|
|
||||||
`default: all` switches on until the canonical copy disables them.
|
|
||||||
|
|
||||||
- **`script/bootstrap` installs a pinned tool by comparing versions, never by
|
|
||||||
testing presence.** An `if ! command -v <tool>; then install; fi` guard tests
|
|
||||||
`PATH` only, so on an already-provisioned machine the pin is inert and a
|
|
||||||
version bump is a silent no-op — while the Dockerfile, installing into a clean
|
|
||||||
image, gets the pinned version, so a local `make check` and `make docker` can
|
|
||||||
disagree about what the tool even is. The canonical form:
|
|
||||||
- compares the installed version against the pin over the **whole** version
|
|
||||||
token; a parser that stops at the first `-` reports `2.12.2` for a host
|
|
||||||
running `2.12.2-rc1` and skips the install;
|
|
||||||
- treats absent, non-zero, empty or unrecognised `--version` output as a
|
|
||||||
mismatch, so the failure direction is a redundant install and never a
|
|
||||||
skipped one;
|
|
||||||
- after installing, re-resolves the binary the way callers do — `hash -r`,
|
|
||||||
then through `PATH`, not through the directory the installer wrote to —
|
|
||||||
and fails naming the resolved path, since an install that a shadowing
|
|
||||||
binary hides succeeds while changing nothing any caller sees;
|
|
||||||
- is actually called, and prints the version on both success paths: a
|
|
||||||
function defined and never invoked has the same exit status and the same
|
|
||||||
empty output as one that worked.
|
|
||||||
|
|
||||||
Keep it POSIX sh: no arrays, no `[[`, no `grep -P`.
|
|
||||||
|
|
||||||
A Go tool a repo needs on the host is installed with `go install` pinned to
|
|
||||||
a commit hash (`go install <package>@<commit hash>`). It is never tracked as
|
|
||||||
a `go.mod` tool dependency or through a `tools.go` file, either of which
|
|
||||||
pulls the tool's own dependencies into the repo's `go.mod` and `go.sum`.
|
|
||||||
|
|
||||||
- When pinning images or packages by hash, add a comment above the reference
|
- When pinning images or packages by hash, add a comment above the reference
|
||||||
with the version and date (YYYY-MM-DD).
|
with the version and date (YYYY-MM-DD).
|
||||||
@@ -665,21 +371,15 @@ style conventions are in separate documents:
|
|||||||
Never edit existing migrations after release.
|
Never edit existing migrations after release.
|
||||||
|
|
||||||
- All repos should have an `.editorconfig` enforcing the project's indentation
|
- All repos should have an `.editorconfig` enforcing the project's indentation
|
||||||
settings: the standard file from
|
settings.
|
||||||
`https://git.eeqj.de/sneak/prompts/raw/branch/main/.editorconfig`, which sets
|
|
||||||
tabs for `Makefile` and Go files, followed by the repo's own sections, such as
|
|
||||||
one for another language it uses. A re-vendor replaces the standard part and
|
|
||||||
keeps those sections.
|
|
||||||
|
|
||||||
- Avoid putting files in the repo root unless necessary. Root should contain
|
- Avoid putting files in the repo root unless necessary. Root should contain
|
||||||
only project-level config files (`README.md`, `AGENTS.md`, `Makefile`,
|
only project-level config files (`README.md`, `Makefile`, `Dockerfile`,
|
||||||
`Dockerfile`, `LICENSE`, `.gitignore`, `.editorconfig`, `REPO_POLICIES.md`,
|
`LICENSE`, `.gitignore`, `.editorconfig`, `REPO_POLICIES.md`, and
|
||||||
and language-specific config). Everything else goes in a subdirectory.
|
language-specific config). Everything else goes in a subdirectory. Canonical
|
||||||
Canonical subdirectory names:
|
subdirectory names:
|
||||||
- `bin/` — executable scripts and tools
|
- `bin/` — executable scripts and tools
|
||||||
- `cmd/` — Go command entrypoints; thin only: one `main.go` per binary whose
|
- `cmd/` — Go command entrypoints
|
||||||
body is a single call into `internal/` or `pkg/`, no project logic in
|
|
||||||
`cmd/`
|
|
||||||
- `configs/` — configuration templates and examples
|
- `configs/` — configuration templates and examples
|
||||||
- `deploy/` — deployment manifests (k8s, compose, terraform)
|
- `deploy/` — deployment manifests (k8s, compose, terraform)
|
||||||
- `docs/` — documentation and markdown (README.md stays in root)
|
- `docs/` — documentation and markdown (README.md stays in root)
|
||||||
@@ -706,7 +406,3 @@ style conventions are in separate documents:
|
|||||||
- Go: `go.mod`, `go.sum`, `.golangci.yml`
|
- Go: `go.mod`, `go.sum`, `.golangci.yml`
|
||||||
- JS: `package.json`, `yarn.lock`, `.prettierrc`, `.prettierignore`
|
- JS: `package.json`, `yarn.lock`, `.prettierrc`, `.prettierignore`
|
||||||
- Python: `pyproject.toml`
|
- Python: `pyproject.toml`
|
||||||
|
|
||||||
- Guidance for coding agents lives in one `AGENTS.md` at the repository root. It
|
|
||||||
is never committed under a file or directory named after one agent tool, such
|
|
||||||
as `CLAUDE.md` or `.claude/`, and never split into separate memory files.
|
|
||||||
|
|||||||
@@ -1,373 +1,453 @@
|
|||||||
# Workflow
|
# Workflow
|
||||||
|
|
||||||
- take an issue from the `1.0.0` milestone on the tracker; work not yet on the
|
- take an issue from the `1.0.0` milestone on the tracker; work not
|
||||||
tracker gets filed as an issue first
|
yet on the tracker gets filed as an issue first
|
||||||
- branch from `next`
|
- branch (from `main`)
|
||||||
- do the work, with tests, in small focused commits
|
- do the work, with tests, in small focused commits
|
||||||
- record it at the top of Completed Steps (`TODO.md` changes in the same commit
|
- record it at the top of Completed Steps (`TODO.md` changes in the
|
||||||
as the work)
|
same commit as the work)
|
||||||
- push the branch and open a PR against `next` whose title ends with
|
- push the branch and open a PR whose title ends with
|
||||||
` (closes #N)`
|
` (closes #N)`
|
||||||
- an independent review gates each merge to `next`; every finding is addressed
|
- an independent review gates the merge; every finding is addressed
|
||||||
or explicitly rebutted on the PR
|
or explicitly rebutted on the PR
|
||||||
- only the owner merges `next` to `main`
|
- merge to `main` once the review passes
|
||||||
|
|
||||||
# Status
|
# Status
|
||||||
|
|
||||||
- pre-1.0
|
- pre-1.0
|
||||||
- the Gitea tracker is authoritative for the pre-1.0 backlog: the open issues
|
- the Gitea tracker is authoritative for the pre-1.0 backlog: the
|
||||||
under the `1.0.0` milestone are what remains before the tag, and this file
|
open issues under the `1.0.0` milestone are what remains before
|
||||||
records history and process, not the queue
|
the tag, and this file records history and process, not the queue
|
||||||
|
|
||||||
# Next Step
|
# Next Step
|
||||||
|
|
||||||
- take the next issue from the `1.0.0` milestone on the tracker:
|
- take the next issue from the `1.0.0` milestone on the tracker:
|
||||||
https://git.eeqj.de/sneak/sfdupes/milestone/17 — the milestone is the source
|
https://git.eeqj.de/sneak/sfdupes/milestone/17 — the milestone is
|
||||||
of truth for what is left before 1.0.0. Individual issues are deliberately not
|
the source of truth for what is left before 1.0.0. Individual
|
||||||
restated here; a copy in this file drifts out of date the moment the tracker
|
issues are deliberately not restated here; a copy in this file
|
||||||
moves
|
drifts out of date the moment the tracker moves
|
||||||
|
|
||||||
# Completed Steps
|
# Completed Steps
|
||||||
|
|
||||||
- re-vendor the canonical files and model scripts from `sneak/prompts` `next` at
|
- replace the 1 KiB end-window sampling with the head/tail plus
|
||||||
`c55a0cb`: golangci-lint v2.14.0; lint and test are phases of the `Dockerfile`
|
content-hash ladder (2026-09-22, branch `next`, closes
|
||||||
that write no image, and `make test` runs the suite under the race detector,
|
https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is
|
||||||
so `Dockerfile.lint`, `script/verify-lint-image-pin` and `make test-race` are
|
hashed in full and compared directly, with no end-window step — its
|
||||||
gone; every `docker build` in `script/` passes `--no-cache`; prettier runs on
|
`head`, `tail`, and `content` all hold the whole-file hash. A file at
|
||||||
the host, from the node and yarn `script/bootstrap` installs, so the
|
10 MiB or above gets only the 64 KiB `head` and `tail` in the hash
|
||||||
`prettier` and `markdown` stages are gone and the build stage installs `git`
|
phase; a new content phase, after the update phase, reads it for its
|
||||||
and `make` itself; a new push cancels the workflow's older run on the same
|
`content` hash — the whole file below 50 MiB, gigabyte-spaced 1 MiB
|
||||||
branch, and a run stops after 20 minutes; `.claude/settings.json` is deleted
|
samples at or above — only when its size, `head`, and `tail` match
|
||||||
(2026-10-08, https://git.eeqj.de/sneak/sfdupes/issues/95)
|
another record's, from the same scan or stored by an earlier one, so
|
||||||
|
a stored file gains its content hash when it gains a match. A file
|
||||||
|
that is gone or has changed since its record was written is not
|
||||||
|
read. The `content` column is part of the version 1 schema. `report`
|
||||||
|
and `trees` group by the extended signature and leave out any record
|
||||||
|
without a `content` hash, so the ladder is applied across the whole
|
||||||
|
database. README "Duplicate detection" documents every rung including
|
||||||
|
the probabilistic large-file path.
|
||||||
|
|
||||||
- `scan` records mtime to the nanosecond, as whole seconds in `mtime` plus
|
- remove the dead `files.dat` references from `Makefile`, `.gitignore`
|
||||||
`mtime_nsec`, and compares it at that resolution, so a same-size rewrite
|
and `.dockerignore` (2026-09-21, branch `next`, closes
|
||||||
within the same second is re-hashed (2026-10-07,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/12)
|
|
||||||
|
|
||||||
- cut the narration from `TODO.md` Completed Steps and from the comments in
|
|
||||||
`script/`, `Dockerfile` and `Dockerfile.lint` (gone since
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/95); §Workflow now branches from and
|
|
||||||
merges to `next` (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/49)
|
|
||||||
|
|
||||||
- `make test-race` ran the test suite under the race detector in a cgo-enabled
|
|
||||||
container, outside `make check` (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/18). Since
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/95 `make test` itself runs the suite
|
|
||||||
under the race detector, in the `Dockerfile`'s Debian-based `test` phase, and
|
|
||||||
`make test-race` is gone
|
|
||||||
|
|
||||||
- a bare `docker build .` failed, naming `script/cibuild` and `script/docker`,
|
|
||||||
rather than serve the gates from cache (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/39). Since
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/95 a bare build succeeds, as
|
|
||||||
`REPO_POLICIES.md` requires, and may serve the gates from cache; the builds in
|
|
||||||
`script/` pass `--no-cache`, so theirs always run
|
|
||||||
|
|
||||||
- `make fmt` and `make fmt-check` run prettier over all Markdown, and CI checks
|
|
||||||
it; all Markdown reformatted (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/19)
|
|
||||||
|
|
||||||
- `script/lint` writes no image, so a run no longer leaves an untagged one
|
|
||||||
behind (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/48)
|
|
||||||
|
|
||||||
- tests cover a missing database, `scan` keeping stdout empty, its skip warning,
|
|
||||||
the `report` and `trees` summary lines, and every subcommand going through
|
|
||||||
`runE` (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/16)
|
|
||||||
|
|
||||||
- `.golangci.yml` replaced with the current canonical copy, which uses
|
|
||||||
`gomodguard_v2`, so lint no longer prints a deprecation warning (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/26)
|
|
||||||
|
|
||||||
- a test fails when either `hashWorker` cancellation check in `scan.go` is
|
|
||||||
removed (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/83)
|
|
||||||
|
|
||||||
- a database path holding `?`, `#` or `%` opens exactly the file it names
|
|
||||||
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/55)
|
|
||||||
|
|
||||||
- `scan` rejects `--workers` below 1 as a usage error instead of running
|
|
||||||
single-threaded (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/10)
|
|
||||||
|
|
||||||
- a test fails when either walk cancellation check in `scan.go` is removed
|
|
||||||
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/81)
|
|
||||||
|
|
||||||
- test that `scan` refuses a database with another schema version (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/64)
|
|
||||||
|
|
||||||
- correct four inaccurate comments in `cancel_test.go` and rename
|
|
||||||
`walkCancelInFlightDirs` to `walkCancelInFlightFiles` (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/33)
|
|
||||||
|
|
||||||
- test the `-x` filesystem-boundary rules in `subdirJob` (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/17)
|
|
||||||
|
|
||||||
- `scan` creates the schema in one transaction; a version-0 database with a
|
|
||||||
`files` table is refused with a clear schema-version error (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/11)
|
|
||||||
|
|
||||||
- README documents install, Docker, a daily cron scan and how to read and check
|
|
||||||
the reports (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/54)
|
|
||||||
|
|
||||||
- the `Dockerfile` build stage kept the Go module cache out of `builder`'s home
|
|
||||||
and copied the sources with `--chown`, so no `chown -R` walked them
|
|
||||||
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/43). Since
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/95 there is no `builder` user and
|
|
||||||
nothing changes owner: the build stage only compiles, as root, and the tests
|
|
||||||
run as `nobody` in the `test` phase
|
|
||||||
|
|
||||||
- `--version` prints `sfdupes VERSION` to stdout; README documents it and
|
|
||||||
`--help` (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/15)
|
|
||||||
|
|
||||||
- `scan` stops cleanly on `SIGINT` or `SIGTERM`: commits what it has hashed,
|
|
||||||
deletes nothing more, exits 1 (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/5)
|
|
||||||
|
|
||||||
- `report` and `trees` stream the records instead of holding them all in memory;
|
|
||||||
the schema gains the `files_signature` index (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/14)
|
|
||||||
|
|
||||||
- progress prints at once on a non-terminal, uses a real terminal test, and
|
|
||||||
prints warnings through a spinner instead of racing its redraw (2026-10-03,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/13)
|
|
||||||
|
|
||||||
- warn about and skip symlink, socket, FIFO, device and `.zfs` operands, keeping
|
|
||||||
the records beneath them (2026-10-03,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/9)
|
|
||||||
|
|
||||||
- `scan` holds a lock on a lock file beside the database for its whole run, so a
|
|
||||||
second `scan` fails at once with exit 1 (2026-10-03,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/53)
|
|
||||||
|
|
||||||
- test stdout write failures in `report` and `trees`; README states that
|
|
||||||
`| head` ends sfdupes by `SIGPIPE` and `>&-` writes to `/dev/null`
|
|
||||||
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/30)
|
|
||||||
|
|
||||||
- `report` and `trees` open the database read-only, and `scan` leaves it out of
|
|
||||||
WAL mode, so reading needs only read access (2026-10-03, closes
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/8)
|
|
||||||
|
|
||||||
- escape tabs, newlines, carriage returns and backslashes in report, trees and
|
|
||||||
warning paths; the root directory's path is `/` (2026-10-03,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/7)
|
|
||||||
|
|
||||||
- stamp the git tag or short commit in a plain `docker build .` instead of `dev`
|
|
||||||
(2026-10-02, branch `next`, closes
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` sends `.git`
|
|
||||||
without `.git/config`; the build stage stamps the `VERSION` build argument,
|
|
||||||
else `git describe --tags --always`, and fails if the context carries `.git`
|
|
||||||
and the version is still empty, `dev` or `unknown`. CI checks out the full
|
|
||||||
history (`fetch-depth: 0`) so it stamps the same value as `make build`.
|
|
||||||
|
|
||||||
- replace the 1 KiB end-window sampling with the head/tail plus content-hash
|
|
||||||
ladder (2026-09-22, branch `next`, closes
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/61); README "Duplicate detection"
|
|
||||||
documents every rung. A file under 10 MiB is hashed in full, and its `head`,
|
|
||||||
`tail` and `content` all hold that hash. A larger file gets only its 64 KiB
|
|
||||||
`head` and `tail` in the hash phase; the content phase, after the update
|
|
||||||
phase, reads it for `content` (the whole file below 50 MiB, gigabyte-spaced 1
|
|
||||||
MiB samples at or above) only when its size, `head` and `tail` match another
|
|
||||||
record's from this scan or an earlier one, and never reads a file gone or
|
|
||||||
changed since its record was written. `report` and `trees` leave out any
|
|
||||||
record without a `content` hash. The `content` column is part of the version 1
|
|
||||||
schema.
|
|
||||||
|
|
||||||
- remove the dead `files.dat` references from `Makefile`, `.gitignore` and
|
|
||||||
`.dockerignore` (2026-09-21, branch `next`, closes
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/22)
|
https://git.eeqj.de/sneak/sfdupes/issues/22)
|
||||||
|
|
||||||
- fix the lint-image pin comments and `FROM` form in `Dockerfile` and
|
- fix the lint-image pin comments and `FROM` form in `Dockerfile` and
|
||||||
`Dockerfile.lint` (2026-08-10, branch `next`, closes
|
`Dockerfile.lint` (2026-08-10, branch `next`, closes
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/25): both pins became the policy
|
https://git.eeqj.de/sneak/sfdupes/issues/25): dropped the false
|
||||||
`# image:vX.Y.Z, YYYY-MM-DD` comment over a bare `FROM image@sha256:...`,
|
`(Debian-based)` parenthetical (v2.12.1 was Debian too) and the
|
||||||
without the false `(Debian-based)` note or the tag. Since
|
redundant tag, so both pins are the policy `# image:vX.Y.Z,
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/95 the `Dockerfile`'s `lint` phase
|
YYYY-MM-DD` comment over a bare `FROM image@sha256:...`. Digest
|
||||||
holds the only golangci-lint pin, in that form, so `Dockerfile.lint` and
|
unchanged. `script/verify-lint-image-pin` parses those `FROM` lines
|
||||||
`script/verify-lint-image-pin`, which compared the two pins, are gone.
|
and still matches the tagless form; its advice line lost the now
|
||||||
|
meaningless "tag and digest". With no tag in either reference, a
|
||||||
|
tag-only disagreement no longer exists — a one-sided tag is caught as
|
||||||
|
a plain mismatch.
|
||||||
|
|
||||||
- run all linting in Docker via `Dockerfile.lint` and `script/lint` (2026-08-10,
|
- run all linting in Docker via `Dockerfile.lint` and `script/lint`
|
||||||
branch `next`, closes https://git.eeqj.de/sneak/sfdupes/issues/46): per the
|
(2026-08-10, branch `next`, closes
|
||||||
owner ruling the linter is never installed on a host, and
|
https://git.eeqj.de/sneak/sfdupes/issues/46): per the owner ruling, the
|
||||||
`golangci-lint config verify` runs before `golangci-lint run`.
|
linter runs inside a container invoked through the `script/`
|
||||||
`script/bootstrap` stopped installing or pinning the linter, and
|
entrypoint and is never installed on a host. New root
|
||||||
`script/verify-linter-pin` was retired. Since
|
`Dockerfile.lint` COPYs the repo into the digest-pinned
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/95 both commands run in the
|
`golangci/golangci-lint:v2.12.2` image and runs
|
||||||
`Dockerfile`'s `lint` phase, which `script/lint` builds alone and the build
|
`golangci-lint config verify` and `golangci-lint run` as build
|
||||||
stage depends on through `COPY --from=lint /src/go.sum /dev/null`;
|
steps, so a successful build IS a clean lint; `script/lint` is
|
||||||
`Dockerfile.lint` and `script/verify-lint-image-pin` are gone. Nothing inside
|
reduced to building it. `script/bootstrap` loses the `go install`,
|
||||||
an image build may run docker, so the phase calls `golangci-lint` directly.
|
the pin constants, the version parser and `verify_golangci_lint`
|
||||||
|
outright rather than hardening them — with nothing linting on the
|
||||||
|
host, the `$GOPATH/bin` versus `PATH` problem that motivated them has
|
||||||
|
no subject — and now warns rather than fails when `docker` is absent.
|
||||||
|
Two traps handled. A lint build on an unchanged tree returns success
|
||||||
|
in well under a second having run no linter, which is
|
||||||
|
https://git.eeqj.de/sneak/sfdupes/issues/32 and
|
||||||
|
https://git.eeqj.de/sneak/sfdupes/issues/39 again, so
|
||||||
|
`Dockerfile.lint` carries `ARG CHECK_EPOCH` referenced
|
||||||
|
inside every gate `RUN` (BuildKit hashes the expanded command, not
|
||||||
|
the declaration) and `script/lint` passes `"$(date +%s)-$$"` — the
|
||||||
|
PID matters because two lint runs land inside the same second easily.
|
||||||
|
And nothing inside an image build may shell out to docker, so the
|
||||||
|
main `Dockerfile`'s lint stage now invokes `golangci-lint` directly
|
||||||
|
instead of `make lint`, and its build stage runs `make test` and
|
||||||
|
`make fmt-check` instead of the `make check` aggregate (`make`, not
|
||||||
|
the scripts bare, because the Makefile's `export CGO_ENABLED = 0`
|
||||||
|
only reaches what it invokes). `COPY --from=lint`
|
||||||
|
`/usr/bin/golangci-lint` is replaced by
|
||||||
|
`COPY --from=lint /src/go.sum /dev/null`: the copied binary was the
|
||||||
|
only edge forcing BuildKit to finish linting before the build stage
|
||||||
|
starts, and dropping it without replacing the edge would have ended
|
||||||
|
fail-fast linting silently under a still-green build. That is
|
||||||
|
canonical `REPO_POLICIES.md:107`'s ordering edge, restored.
|
||||||
|
`ENV PATH=/home/builder/go/bin:$PATH` is gone with the `go install`
|
||||||
|
that justified it. `script/verify-linter-pin` is retired, deleted
|
||||||
|
along with its README entry, because both of its subjects ceased to
|
||||||
|
exist in the same change: it compared a linter binary against
|
||||||
|
`GOLANGCI_LINT_VERSION` in `script/bootstrap`, and there is now
|
||||||
|
neither a binary crossing between stages nor a version pin in
|
||||||
|
bootstrap. The drift it guarded has not gone away, it has moved — the
|
||||||
|
linter is still pinned twice, now as the `FROM` line of
|
||||||
|
`Dockerfile.lint` and the `FROM` line of the `Dockerfile` lint stage,
|
||||||
|
with nothing syncing them, which is exactly what
|
||||||
|
https://git.eeqj.de/sneak/sfdupes/issues/42 made a build failure. Its
|
||||||
|
replacement is one new `script/verify-lint-image-pin`,
|
||||||
|
run as a gate in both files, which compares the two references to
|
||||||
|
each other and deliberately restates neither: a hardcoded expected
|
||||||
|
digest would be a third copy and the same drift one file further out.
|
||||||
|
`golangci-lint config verify` is included per the ruling, and the
|
||||||
|
concern about its unpinned live HTTPS schema fetch was measured
|
||||||
|
rather than assumed — under `--network none` the pinned binary both
|
||||||
|
passes a valid config and rejects an invalid one with the jsonschema
|
||||||
|
error, so it validates from an embedded schema and makes no network
|
||||||
|
call of its own. The README scopes that to the gate steps rather
|
||||||
|
than to linting as a whole: `Dockerfile.lint` runs `go mod download`
|
||||||
|
above them, so a cold cache still needs the network and only a warm
|
||||||
|
one lints offline. Verified: `make lint` green with every `PATH`
|
||||||
|
directory containing a `golangci-lint` removed
|
||||||
|
(`/home/user/go/bin`, `/home/user/.local/bin`, `/usr/local/bin`;
|
||||||
|
`command -v golangci-lint` empty); two consecutive `script/lint` runs
|
||||||
|
on an untouched tree both executed the linter, 27.7s and 28.7s in the
|
||||||
|
lint step under distinct epochs with the `COPY . .` layer `CACHED`
|
||||||
|
above them, at 42.2s and 41.8s wall clock — the no-cache rule was not
|
||||||
|
weakened to shorten that. Negative control: a planted
|
||||||
|
`var unusedIssue46Sentinel = 1` failed `script/lint` with
|
||||||
|
`report.go:173:5: var unusedIssue46Sentinel is unused (unused)`, and
|
||||||
|
failed `make docker` at `[lint 9/9]` with the build stage stopped at
|
||||||
|
`[builder 3/12]` — `COPY --from=lint`, `script/bootstrap`, the test
|
||||||
|
gate and `make build` all zero occurrences — then reverted clean. The
|
||||||
|
drift guard fails on a tag-only disagreement, on a digest-only
|
||||||
|
disagreement, and on an unreadable reference, naming both sides.
|
||||||
|
`make docker` green in 5m35s with all six gates executing under one
|
||||||
|
epoch (lint 37.6s, test 25.2s reporting
|
||||||
|
`ok sneak.berlin/go/sfdupes 1.938s coverage: 88.5%`, not `(cached)`).
|
||||||
|
The non-root quirk still holds: in the builder image with the Go test
|
||||||
|
cache off, `--user 0:0` fails `TestScanHardlinkRunFailsTogether`
|
||||||
|
(exit 1) where the unprivileged user passes (exit 0). Noted for
|
||||||
|
follow-up, not fixed here: `golangci-lint` warns that the
|
||||||
|
`gomodguard` linter is deprecated since v2.12.0 in favour of
|
||||||
|
`gomodguard_v2`.
|
||||||
|
|
||||||
- install the Docker build stage's prerequisites by running `script/bootstrap`
|
- install the Docker build stage's prerequisites by running
|
||||||
instead of `apk add --no-cache make` inline (2026-08-09, branch
|
`script/bootstrap` instead of `apk add --no-cache make` inline
|
||||||
`dockerfile-bootstrap`, closes https://git.eeqj.de/sneak/sfdupes/issues/42),
|
(2026-08-09, branch `dockerfile-bootstrap`, closes #42): canonical
|
||||||
with `script/verify-linter-pin` failing the build unless the linter copied
|
`REPO_POLICIES.md:97` requires it, and the inline install left the
|
||||||
from the lint stage was the version `script/bootstrap` pinned. Builds then
|
build stage maintaining its own notion of the toolchain — exactly
|
||||||
took up to 5m14s cold, mostly in a `chown -R` of the module cache, filed as
|
the divergence #24 exists to close, one layer down. The stage now
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/43. Since
|
copies `script/` plus `go.mod`/`go.sum` and runs `script/bootstrap`,
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/95 the build stage installs `git` and
|
which ends in `go mod download`, so the separate invocation of that
|
||||||
`make` with `apk add --no-cache` and only compiles; the `lint` and `test`
|
is gone. `COPY --from=lint /usr/bin/golangci-lint` stays, and moves
|
||||||
phases are the gates
|
above the bootstrap layer. It is the only edge making this stage
|
||||||
- bust the Docker layer cache for the gate steps, so `script/cibuild` and
|
depend on the lint stage, so deleting it as redundant would end
|
||||||
`script/docker` cannot report a green they did not earn (2026-08-09, branch
|
fail-fast linting silently. Letting bootstrap install its own linter
|
||||||
`cibuild-cache-bust`, closes https://git.eeqj.de/sneak/sfdupes/issues/32): the
|
here would have reintroduced the second toolchain and paid for a
|
||||||
`Dockerfile` copies the tree before its gates, so on an unchanged tree Docker
|
from-source build of it. What makes the two stages provably one
|
||||||
served them from cache and the build exited 0 having run nothing. The fix was
|
toolchain rather than two that happen to agree is a new
|
||||||
a `CHECK_EPOCH` build argument; since
|
`script/verify-linter-pin`, run in the build stage on the binary
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/95 every `docker build` in `script/`
|
that arrives from the lint stage, before bootstrap: it fails the
|
||||||
passes `--no-cache` instead. Run as root, the tests fail
|
build naming both versions unless that binary is the version
|
||||||
`TestScanHardlinkRunFailsTogether`, because root reads through the `chmod(0)`
|
`script/bootstrap` pins. Bootstrap's own check could not serve that
|
||||||
the test relies on, so the `test` phase runs them as `nobody`
|
purpose — it reinstalls its pin from source and then verifies
|
||||||
- check the installed golangci-lint version in `script/bootstrap` instead of
|
whatever `PATH` resolves, so drift self-heals silently and a lint
|
||||||
only its presence (2026-08-09, branch `bootstrap-version-check`, closes
|
stage image bumped on its own would lint at the new version while
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/24): the version lives only in
|
`make check` ran at the old one, green. The linter version is pinned
|
||||||
`GOLANGCI_LINT_VERSION`, with the `go install` module ref derived from it, and
|
in two independent places (the lint stage image digest and
|
||||||
any installed version that is not the pin — older, newer, absent or
|
`GOLANGCI_LINT_VERSION`) and nothing else keeps them in sync, so a
|
||||||
unparseable — is reinstalled. `go install` writes into `GOBIN` (or
|
half-applied bump is now a build failure. The pin is read out of
|
||||||
`GOPATH/bin`) while `make lint` runs the first `golangci-lint` on `PATH`, so
|
`script/bootstrap`, which stays the single source of truth; a pin
|
||||||
bootstrap re-reads the effective version after installing and, on a mismatch,
|
that cannot be read is a hard failure, not a skip. The check needs
|
||||||
prints both paths and both versions and exits non-zero; it does not reorder
|
no `CHECK_EPOCH`: its only inputs are the copied binary and
|
||||||
`PATH` or delete anyone's binary. The `--version` call keeps its stderr and is
|
`script/`, so Docker invalidates the layer exactly when a cached
|
||||||
bounded by `timeout(1)` where that exists. `git`, `make` and `go` keep
|
result would stop being true, and it is documented with the other
|
||||||
presence-only checks. Verified by bootstrapping this host from v2.10.1 to
|
entrypoints in the README. `$GOPATH/bin` joins `PATH` because
|
||||||
v2.12.2 and again to a no-op, and by stub runs of the script under `dash`
|
that is where bootstrap's `go install` lands and bootstrap verifies
|
||||||
covering a thirteen-input version-parse matrix, a shadowed install that must
|
its installs against what `PATH` resolves — nothing in the image is
|
||||||
exit non-zero, an install destination not on `PATH`, `GOBIN` set, and a wedged
|
shadowed by it, the directory does not exist until bootstrap runs.
|
||||||
binary that must hit the timeout; `make check` and `make lint` are clean at
|
Everything added sits above `ARG CHECK_EPOCH`, and the `chown` and
|
||||||
v2.12.2, so v2.10.1 was not hiding any findings on `main`
|
`USER builder` still precede `make check`. Verified: the guard fails
|
||||||
|
the build with both versions named when the lint stage's linter is
|
||||||
|
faked to a different version, and an unmodified build still passes
|
||||||
|
it; bootstrap runs clean under Alpine's `sh` and its `apk` branch,
|
||||||
|
installing `git` and `make` and finding the copied
|
||||||
|
linter already at the pin; a second build served the bootstrap and
|
||||||
|
dependency layers `CACHED` while both gates ran with a fresh epoch;
|
||||||
|
a planted `unused` finding failed the build at the lint gate in
|
||||||
|
48.9s with the build stage's `make check` never starting; and the
|
||||||
|
suite run in the image as `--user 0:0` fails
|
||||||
|
`TestScanHardlinkRunFailsTogether`, so the drop to the unprivileged
|
||||||
|
user is still load-bearing. That last check needs the Go test cache
|
||||||
|
disabled — the first attempt reported `ok ... (cached)` as root,
|
||||||
|
reusing the result the build-time run had left in the shared cache,
|
||||||
|
which would have read as a pass. Build wall time, on a shared host
|
||||||
|
running many concurrent builds and so noisy: 2m13s on an unchanged
|
||||||
|
tree, 2m17s and 4m29s for two builds after a source change, 5m14s
|
||||||
|
cold. Only the cold one breaches the policy ceiling, and not because
|
||||||
|
of this change — `chown -R builder:builder /src /home/builder` walks
|
||||||
|
the module cache and re-runs on every source change, and it alone
|
||||||
|
varied between 77s and 210s across those four builds, which is also
|
||||||
|
the whole spread in the totals. The same cold measurement against
|
||||||
|
`main` is 5m03s with a 209s `chown`. Filed as #43
|
||||||
|
- bust the Docker layer cache for the gate steps, so `script/cibuild`
|
||||||
|
and `script/docker` cannot report a green they did not earn
|
||||||
|
(2026-08-09, branch `cibuild-cache-bust`, closes #32): both scripts
|
||||||
|
were bare `docker build` invocations with no cache control, and the
|
||||||
|
`Dockerfile` copies the tree before running its gates, so on an
|
||||||
|
unchanged tree Docker served those layers from cache and the build
|
||||||
|
exited 0 having executed nothing. That is not hypothetical here —
|
||||||
|
every merge this repo has done is a non-fast-forward merge of an
|
||||||
|
undiverged branch, so each merge commit's tree is byte-identical to
|
||||||
|
the branch head's and each merge CI run was almost certainly a full
|
||||||
|
cache hit; and PR #31's reviewer found `make docker` returning
|
||||||
|
success as a 17-layer cache hit, catching it only by being
|
||||||
|
suspicious. The fix is `ARG CHECK_EPOCH` with the scripts passing
|
||||||
|
`--build-arg CHECK_EPOCH="$(date +%s)"`. Two details make or break
|
||||||
|
it. `ARG` is scoped per stage and this `Dockerfile` has three gates
|
||||||
|
across two — `make fmt-check` and `make lint` in the lint stage,
|
||||||
|
`make check` in the build stage — so a single declaration would have
|
||||||
|
left one stage silently cacheable; it is declared in both. And
|
||||||
|
BuildKit hashes the expanded command, not the declaration, so a
|
||||||
|
declared-but-unreferenced `ARG` invalidates nothing: each gate `RUN`
|
||||||
|
echoes the epoch, which also puts the value in the build log as
|
||||||
|
evidence the layer really ran. Placement is below the dependency
|
||||||
|
layers on purpose — a build that goes cold every time would be a
|
||||||
|
different bug, not a fix. Verified by running each script twice back
|
||||||
|
to back on an unchanged tree under `BUILDKIT_PROGRESS=plain`: all
|
||||||
|
three gates executed on all four runs, each with a fresh epoch in
|
||||||
|
the log (`script/cibuild` 78.8s then 61.1s; `script/docker` 61.1s
|
||||||
|
then 53.4s), and twelve steps were still served `CACHED` in the
|
||||||
|
steady state — both `go mod download`s, `apk add`, `adduser`, the
|
||||||
|
`chown`, every `go.mod`/`go.sum` and source copy, the linter copy
|
||||||
|
out of the lint stage, and the binary copy into the runtime stage.
|
||||||
|
The lint stage still gates the build stage: with a deliberate
|
||||||
|
`unused` finding planted in the tree, the build failed at
|
||||||
|
`make lint` in 36.1s and the build-stage `make check` never started.
|
||||||
|
The build stage also still drops to the unprivileged `builder` user
|
||||||
|
before `make check`, which the suite depends on rather than merely
|
||||||
|
prefers: forcing the same image to run the tests as root fails
|
||||||
|
`TestScanHardlinkRunFailsTogether`, because root reads straight
|
||||||
|
through the `chmod(0)` the test uses to prove hard links are read
|
||||||
|
once. This is the local fix only; propagating it to the canonical
|
||||||
|
templates is `prompts` #26
|
||||||
|
- check the installed golangci-lint version in `script/bootstrap`
|
||||||
|
instead of only its presence (2026-08-09, branch
|
||||||
|
`bootstrap-version-check`, closes #24): `missing golangci-lint` meant
|
||||||
|
any linter already on `PATH` satisfied the check, so the pin was never
|
||||||
|
consulted and the v2.12.2 bump from #3 was inert on every host that
|
||||||
|
already had one — this host ran v2.10.1 against a v2.12.2 pin,
|
||||||
|
`make check` went green, and `make docker` then rejected the same
|
||||||
|
commit with findings the local gate never saw. The version now lives
|
||||||
|
in one place, `GOLANGCI_LINT_VERSION`, with the `go install` module
|
||||||
|
ref derived from it so a bump cannot half-apply; a
|
||||||
|
`golangci_lint_version` helper parses `golangci-lint --version`
|
||||||
|
(taking the field after the word `version` and tolerating an optional
|
||||||
|
leading `v`, which the module ref carries and the binary's output does
|
||||||
|
not), and any version that is not the pin — older, newer, absent or
|
||||||
|
unparseable — is reinstalled. The install is then verified against the
|
||||||
|
binary `PATH` actually resolves: `go install` writes into `GOBIN` (or
|
||||||
|
`GOPATH/bin`) while `make lint` runs whichever `golangci-lint` comes
|
||||||
|
first on `PATH`, so a wrong-version one sitting ahead of it — nix,
|
||||||
|
apt, brew, apk, or the `/usr/local/bin` copy the `Dockerfile` builder
|
||||||
|
stage makes — would swallow the install and leave the local gate
|
||||||
|
disagreeing with CI under an affirmative `bootstrap complete`.
|
||||||
|
Bootstrap now re-reads the effective version after installing and, on
|
||||||
|
a mismatch, prints both paths and both versions to stderr and exits
|
||||||
|
non-zero instead of claiming success; it does not reorder anyone's
|
||||||
|
`PATH` or delete their binary. The `--version` call keeps its stderr
|
||||||
|
connected, so a present-but-broken binary says why rather than
|
||||||
|
reinstalling forever in silence, and is bounded by `timeout(1)` where
|
||||||
|
that exists, so a wedged binary cannot hang bootstrap. `git`, `make`
|
||||||
|
and `go` keep their presence-only checks and now say why in a
|
||||||
|
comment: they are host package-manager tools the repo deliberately
|
||||||
|
does not pin, with `go.mod` governing the language version and the
|
||||||
|
digest-pinned images covering reproducible builds. Verified on this
|
||||||
|
host by bootstrapping from v2.10.1 to v2.12.2 and running it again to
|
||||||
|
a no-op, plus stub runs of the real script under `dash` covering a
|
||||||
|
thirteen-input parse matrix (absent, older, newer, host-style,
|
||||||
|
image-style, leading-`v`, stderr-only, empty, non-zero exit, impostor
|
||||||
|
binary, `(devel)`, trailing `version`), a shadowed install that must
|
||||||
|
exit non-zero, an install destination not on `PATH` at all, `GOBIN`
|
||||||
|
set, and a wedged binary that must hit the timeout; `make check` and
|
||||||
|
`make lint` are clean at v2.12.2, so v2.10.1 was not hiding any
|
||||||
|
findings on `main`
|
||||||
- unwind the hash worker pool on the error path (2026-08-09, branch
|
- unwind the hash worker pool on the error path (2026-08-09, branch
|
||||||
`hash-pool-cleanup`, closes https://git.eeqj.de/sneak/sfdupes/issues/6): the
|
`hash-pool-cleanup`, closes #6): `hashPhase` used to return the
|
||||||
pool is now an owned, context-aware `hashPool`: every blocking send in the
|
moment `recordRun` failed and abandon the pool — the feeder parked
|
||||||
feeder and the workers selects on `ctx.Done()`, `jobs` is closed on every path
|
forever on a full `jobs` channel and every worker on a full
|
||||||
out, and `hashPhase` defers `pool.stop()`, which cancels and then drains
|
`results` channel. That only stopped being invisible when #4 landed
|
||||||
`results` until the last goroutine has exited — draining is what frees a
|
and `runScan` began unwinding instead of calling `os.Exit`. The
|
||||||
worker already parked on a send. `ctx` is threaded from `cmd.Context()`
|
pool is now an owned, context-aware `hashPool`: every blocking send
|
||||||
through `runScan`, `syncScan`, both worker pools and the whole database layer,
|
in the feeder and the workers selects on `ctx.Done()`, `jobs` is
|
||||||
as the first parameter everywhere. The walk pool gets the same treatment plus
|
closed on every path out, and `hashPhase` defers `pool.stop()`,
|
||||||
a `ctx.Err()` guard after the walk: a cancelled walk yields a partial size
|
which cancels and then drains `results` until the last goroutine
|
||||||
census, and every file it never reached looks vanished to the update phase.
|
has exited — draining is what frees a worker already parked on a
|
||||||
That phase's own `BeginTx` also fails on the cancelled context before deleting
|
send. `ctx` is threaded from `cmd.Context()` through `runScan`,
|
||||||
anything, but the guard is the barrier that still holds once an interrupted
|
`syncScan`, both worker pools and the whole database layer (it is
|
||||||
scan may commit what it has. Tests drive `run(scan)` against a database whose
|
the first parameter everywhere), so #5 can hand this path a signal
|
||||||
insert trigger aborts and assert that the scan fails instead of hanging and
|
and needs to add nothing else. The walk pool never leaked, because
|
||||||
that `runtime.NumGoroutine()` polls back to its pre-scan baseline; others
|
`walkPhase` always drains its events to close, but it has the same
|
||||||
cancel a scan part-way through the walk, deterministically, by counting its
|
unbounded-send shape and #5 will give it an early return, so it
|
||||||
own consultations of `ctx.Done()`, and assert that it stops at the guard
|
gets the same treatment plus a `ctx.Err()` guard after the walk: a
|
||||||
holding a partial census and a still-populated record index, with every record
|
cancelled walk yields a partial size census, and every file it never
|
||||||
intact. Direct tests of `sendEvent`, the walk workers, `dispatchDirs`,
|
reached looks vanished to the update phase. That phase's own
|
||||||
`feedHashJobs`, `hashWorker` and `hashPhase` cover the remaining cancellation
|
`BeginTx` fails on the same cancelled context before deleting
|
||||||
branches of both pools
|
anything, so the guard is defence in depth rather than the only
|
||||||
- guarantee the database is closed on every fatal exit path (2026-08-09, branch
|
barrier — but it is the one that survives #5 deciding an interrupted
|
||||||
`db-close-on-fatal`, closes https://git.eeqj.de/sneak/sfdupes/issues/4):
|
scan may commit what it has. Tests drive `run(scan)` against a
|
||||||
`fatalf` and its `os.Exit(1)` are gone, so the deferred `db.Close()` — and
|
database whose insert trigger aborts, and assert both that the scan
|
||||||
with it the SQLite WAL checkpoint — now actually runs when a subcommand fails;
|
fails instead of hanging and that `runtime.NumGoroutine()` polls
|
||||||
`runScan`, `runReport`, `runTrees`, `loadRecords` and `resolveRoots` return
|
back to its pre-scan baseline; a second set cancels a scan part-way
|
||||||
errors instead. The single exit point is `run` in `main.go`: it maps a
|
through the walk — deterministically, by counting the scan's own
|
||||||
`fatalError` (anything a subcommand returned) to exit 1 and cobra's own
|
consultations of `ctx.Done()` rather than racing a timer — and
|
||||||
argument and flag errors to exit 2, which keeps a runtime failure from being
|
asserts that it stops at the guard holding a partial census and a
|
||||||
reported as a usage error or printing the usage text. New `main_test.go`
|
still-populated record index, with every record intact. The
|
||||||
drives the CLI in-process and asserts the exit codes from README §Error
|
remaining cancellation branches of both pools are covered by direct
|
||||||
handling plus the stdout/stderr split, including that a fatal error raised
|
tests of `sendEvent`, the walk workers, `dispatchDirs`,
|
||||||
after the database is open leaves no `-wal`/`-shm` sidecar behind for `scan`,
|
`feedHashJobs`, `hashWorker` and `hashPhase`
|
||||||
`report` or `trees`
|
- guarantee the database is closed on every fatal exit path
|
||||||
- update golangci-lint to v2.12.2 with the canonical config (2026-08-09, branch
|
(2026-08-09, branch `db-close-on-fatal`, closes #4): `fatalf` and
|
||||||
`golangci-v2.12.2`, merged as `38a01bd`, closes
|
its `os.Exit(1)` are gone, so the deferred `db.Close()` — and with
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/3): bumped the pinned linter in the
|
it the SQLite WAL checkpoint — now actually runs when a subcommand
|
||||||
`Dockerfile` lint stage and `script/bootstrap` from v2.12.1 to v2.12.2, and
|
fails; `runScan`, `runReport`, `runTrees`, `loadRecords` and
|
||||||
replaced `.golangci.yml` with the canonical file — the linter settings (`lll`,
|
`resolveRoots` return errors instead. The single exit point is `run`
|
||||||
`funlen`, `cyclop`, `dupl` thresholds) now live under `linters.settings` per
|
in `main.go`: it maps a `fatalError` (anything a subcommand
|
||||||
the v2 schema, so they are actually applied; no new lint findings surfaced
|
returned) to exit 1 and cobra's own argument and flag errors to exit
|
||||||
- convert Makefile targets to scripts-to-rule-them-all `script/` entrypoints
|
2, which keeps a runtime failure from being reported as a usage
|
||||||
like the other managed repos (2026-07-26, commit `3abeacf`, closes
|
error or printing the usage text. New `main_test.go` drives the CLI
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/1): all 12 `script/` entrypoints
|
in-process and asserts the exit codes from README §Error handling
|
||||||
exist (`bootstrap`, `setup`, `projectname`, `test`, `lint`, `fmt`,
|
plus the stdout/stderr split, including that a fatal error raised
|
||||||
`fmt-check`, `check`, `docker`, `cibuild`, `precommit`, `install-precommit`)
|
after the database is open leaves no `-wal`/`-shm` sidecar behind
|
||||||
and every Makefile target is now a thin shim over them
|
for `scan`, `report` or `trees`
|
||||||
|
- update golangci-lint to v2.12.2 with the canonical config
|
||||||
|
(2026-08-09, branch `golangci-v2.12.2`, merged as `38a01bd`,
|
||||||
|
closes #3): bumped the pinned linter in the `Dockerfile` lint
|
||||||
|
stage and `script/bootstrap` from v2.12.1 to v2.12.2, and replaced
|
||||||
|
`.golangci.yml` with the canonical file — the linter settings
|
||||||
|
(`lll`, `funlen`, `cyclop`, `dupl` thresholds) now live under
|
||||||
|
`linters.settings` per the v2 schema, so they are actually
|
||||||
|
applied; no new lint findings surfaced
|
||||||
|
- convert Makefile targets to scripts-to-rule-them-all `script/`
|
||||||
|
entrypoints like the other managed repos (2026-07-26, commit
|
||||||
|
`3abeacf`, closes #1): all 12 `script/` entrypoints exist
|
||||||
|
(`bootstrap`, `setup`, `projectname`, `test`, `lint`, `fmt`,
|
||||||
|
`fmt-check`, `check`, `docker`, `cibuild`, `precommit`,
|
||||||
|
`install-precommit`) and every Makefile target is now a thin shim
|
||||||
|
over them, matching the other managed repos
|
||||||
- make the binary the default Make target (2026-07-24, branch
|
- make the binary the default Make target (2026-07-24, branch
|
||||||
`make-default-target`): plain `make` now builds `sfdupes` (previously it ran
|
`make-default-target`): plain `make` now builds `sfdupes`
|
||||||
`check` plus `build`); `make build` remains as an alias
|
(previously it ran `check` plus `build`); `make build` remains as
|
||||||
- scan-wide phases, concurrent operands, batched updates (2026-07-24, branch
|
an alias
|
||||||
`scan-wide-phases`): all operands seed the shared walk pool and every pass
|
- scan-wide phases, concurrent operands, batched updates (2026-07-24,
|
||||||
runs once over the whole scan, so totals and ETAs are scan-global; the
|
branch `scan-wide-phases`): all operands seed the shared walk pool
|
||||||
per-operand walk/hash/update cycles and their stderr announcements are gone;
|
and every pass runs once over the whole scan, so totals and ETAs
|
||||||
the update pass commits in batched transactions — the filesystem is
|
are scan-global; the per-operand walk/hash/update cycles and their
|
||||||
authoritative and the database an eventually-consistent reflection, so
|
stderr announcements are gone; the update pass commits in batched
|
||||||
scan-level atomicity is not required
|
transactions — the filesystem is authoritative and the database an
|
||||||
|
eventually-consistent reflection, so scan-level atomicity is not
|
||||||
|
required
|
||||||
- split the stat pass back out of the walk (2026-07-24, branch
|
- split the stat pass back out of the walk (2026-07-24, branch
|
||||||
`parallel-phases`): phases are strictly sequential again — walk, stat, hash,
|
`parallel-phases`): phases are strictly sequential again — walk,
|
||||||
update per operand — with parallelism only inside each phase; the walk
|
stat, hash, update per operand — with parallelism only inside each
|
||||||
enumerates paths with per-directory workers and the stat pass lstats them with
|
phase; the walk enumerates paths with per-directory workers and the
|
||||||
per-file workers, restoring the exact total/ETA stat bar
|
stat pass lstats them with per-file workers, restoring the exact
|
||||||
- announce each operand on stderr before its passes (2026-07-24, branch
|
total/ETA stat bar
|
||||||
`scan-operand-progress`): with per-operand walk/hash/update cycles, a
|
- announce each operand on stderr before its passes (2026-07-24,
|
||||||
multi-operand run (e.g. `scan /srv/*`) showed pass totals that looked like the
|
branch `scan-operand-progress`): with per-operand walk/hash/update
|
||||||
whole run's
|
cycles, a multi-operand run (e.g. `scan /srv/*`) showed pass totals
|
||||||
- parallel walk (2026-07-24, branch `parallel-walk`): the walk pass was a single
|
that looked like the whole run's — an operator watching operand 3 of
|
||||||
goroutine and took hours at ~20M files on a busy pool (observed: 22M files in
|
14 hash 300k files concluded 20M files were being skipped
|
||||||
4h on a ZFS server); it is now a per-directory worker-pool traversal that
|
- parallel walk (2026-07-24, branch `parallel-walk`): the walk pass
|
||||||
records size/mtime during the walk (folding away the separate stat pass,
|
was a single goroutine and took hours at ~20M files on a busy pool
|
||||||
halving metadata I/O), and each `PATH` operand commits in its own transaction
|
(observed: 22M files in 4h on a ZFS server); it is now a
|
||||||
so an interrupted scan keeps completed operands
|
per-directory worker-pool traversal that records size/mtime during
|
||||||
|
the walk (folding away the separate stat pass, halving metadata
|
||||||
|
I/O), and each `PATH` operand commits in its own transaction so an
|
||||||
|
interrupted scan keeps completed operands
|
||||||
|
|
||||||
- persistent scan database (2026-07-24, branch `persistent-database`): `scan`
|
- persistent scan database (2026-07-24, branch `persistent-database`):
|
||||||
now maintains a SQLite database (`modernc.org/sqlite`, pure Go, cgo stays
|
`scan` now maintains a SQLite database (`modernc.org/sqlite`, pure
|
||||||
disabled) keyed by absolute path that survives between runs — a rescan hashes
|
Go, cgo stays disabled) keyed by absolute path that survives between
|
||||||
only new or changed files (by mtime/size), deletes records for files vanished
|
runs — a rescan hashes only new or changed files (by mtime/size),
|
||||||
from under the scanned operands, and leaves records outside them untouched, so
|
deletes records for files vanished from under the scanned operands,
|
||||||
`scan` can be cronned daily; `report` and `trees` read the database (no
|
and leaves records outside them untouched, so `scan` can be cronned
|
||||||
positional arguments) instead of a scan stream. Database at
|
daily; `report` and `trees` read the database (no positional
|
||||||
`/var/lib/sfdupes/db.sqlite`, overridable via `SFDUPES_DATABASE`; WAL
|
arguments) instead of a scan stream. Database at
|
||||||
journaling plus a single-transaction update keep a report run during a scan
|
`/var/lib/sfdupes/db.sqlite`, overridable via `SFDUPES_DATABASE`;
|
||||||
safe
|
WAL journaling plus a single-transaction update keep a report run
|
||||||
- add the `origin` remote (`git@git.eeqj.de:sneak/sfdupes.git`), tag `v0.0.1`,
|
during a scan safe
|
||||||
and push `main` plus tags (2026-07-23)
|
- add the `origin` remote (`git@git.eeqj.de:sneak/sfdupes.git`), tag
|
||||||
|
`v0.0.1`, and push `main` plus tags (2026-07-23)
|
||||||
- `scan` CLI rework (2026-07-23, branch `scan-required-paths`): required
|
- `scan` CLI rework (2026-07-23, branch `scan-required-paths`): required
|
||||||
`PATH...` operands via cobra flags replacing the `/srv` `-root` default; new
|
`PATH...` operands via cobra flags replacing the `/srv` `-root`
|
||||||
`-x`/`--one-file-system` flag (GNU convention) to stop at filesystem
|
default; new `-x`/`--one-file-system` flag (GNU convention) to stop
|
||||||
boundaries, which are crossed by default
|
at filesystem boundaries, which are crossed by default
|
||||||
- bring the repo into full policy compliance (2026-07-23, branch
|
- bring the repo into full policy compliance (2026-07-23, branch
|
||||||
`repo-policy-compliance`; checklist below)
|
`repo-policy-compliance`; checklist below)
|
||||||
- `git init` with README-only first commit; code baseline committed on `main`
|
- `git init` with README-only first commit; code baseline committed on
|
||||||
(2026-07-22)
|
`main` (2026-07-22)
|
||||||
- implement `scan`, `report`, and `trees` subcommands (pre-git history)
|
- implement `scan`, `report`, and `trees` subcommands (pre-git history)
|
||||||
|
|
||||||
# Future Steps
|
# Future Steps
|
||||||
|
|
||||||
- possible later features (explicitly out of scope per README): full-content
|
- possible later features (explicitly out of scope per README):
|
||||||
verification of candidates, removal-script helpers
|
full-content verification of candidates, removal-script helpers
|
||||||
|
|
||||||
# Repo Policy Compliance
|
# Repo Policy Compliance
|
||||||
|
|
||||||
Audited 2026-07-22 against `REPO_POLICIES.md` (2026-07-06), the existing repo
|
Audited 2026-07-22 against `REPO_POLICIES.md` (2026-07-06), the existing
|
||||||
checklist, and the Go styleguide. Code is already gofmt-clean, so no standalone
|
repo checklist, and the Go styleguide. Code is already gofmt-clean, so no
|
||||||
formatting commit is needed.
|
standalone formatting commit is needed.
|
||||||
|
|
||||||
- [x] `.gitignore` missing — the compiled `sfdupes` binary and `files.dat` sit
|
- [x] `.gitignore` missing — the compiled `sfdupes` binary and
|
||||||
untracked in the tree; needs OS/editor/Go artifacts plus secrets patterns
|
`files.dat` sit untracked in the tree; needs OS/editor/Go
|
||||||
|
artifacts plus secrets patterns
|
||||||
- [x] `.editorconfig` missing
|
- [x] `.editorconfig` missing
|
||||||
- [x] `LICENSE` missing and README has no License section (MIT assumed from
|
- [x] `LICENSE` missing and README has no License section (MIT assumed
|
||||||
house convention — user to confirm)
|
from house convention — user to confirm)
|
||||||
- [x] `REPO_POLICIES.md` missing from repo root
|
- [x] `REPO_POLICIES.md` missing from repo root
|
||||||
- [x] `.golangci.yml` missing (install canonical copy); code must then pass
|
- [x] `.golangci.yml` missing (install canonical copy); code must then
|
||||||
`make lint` (150 findings fixed; `make lint` is clean)
|
pass `make lint` (150 findings fixed; `make lint` is clean)
|
||||||
- [x] `Makefile` lacks required targets `test`, `lint`, `fmt`, `fmt-check`,
|
- [x] `Makefile` lacks required targets `test`, `lint`, `fmt`,
|
||||||
`docker`, `hooks`; `check` currently depends on `build`, which writes the
|
`fmt-check`, `docker`, `hooks`; `check` currently depends on
|
||||||
binary (`make check` must not modify files)
|
`build`, which writes the binary (`make check` must not modify
|
||||||
- [x] no tests — `go test ./...` has nothing to run; policy requires real tests
|
files)
|
||||||
with a 30-second timeout and the conditional `-v` rerun pattern (suite
|
- [x] no tests — `go test ./...` has nothing to run; policy requires
|
||||||
covers parsing, grouping, digests, suppression, hashing, and the scan
|
real tests with a 30-second timeout and the conditional `-v`
|
||||||
pipeline; 64% coverage)
|
rerun pattern (suite covers parsing, grouping, digests,
|
||||||
- [x] `Dockerfile` missing — Go multistage with hash-pinned images: fail-fast
|
suppression, hashing, and the scan pipeline; 64% coverage)
|
||||||
lint stage, build stage running `make check`
|
- [x] `Dockerfile` missing — Go multistage with hash-pinned images:
|
||||||
|
fail-fast lint stage, build stage running `make check`
|
||||||
- [x] `.dockerignore` missing
|
- [x] `.dockerignore` missing
|
||||||
- [x] `.gitea/workflows/check.yml` missing (`docker build .` on push, checkout
|
- [x] `.gitea/workflows/check.yml` missing (`docker build .` on push,
|
||||||
action pinned by commit SHA)
|
checkout action pinned by commit SHA)
|
||||||
- [x] README lacks required sections: Description first line
|
- [x] README lacks required sections: Description first line
|
||||||
(name/purpose/category/license/author), Getting Started, Rationale, TODO,
|
(name/purpose/category/license/author), Getting Started,
|
||||||
License, Author
|
Rationale, TODO, License, Author
|
||||||
- [x] README non-goal "no git repository setup and no CI" is stale now that the
|
- [x] README non-goal "no git repository setup and no CI" is stale now
|
||||||
repo is under git with CI
|
that the repo is under git with CI
|
||||||
- [x] pre-commit hook not installed (`make hooks` once the target exists)
|
- [x] pre-commit hook not installed (`make hooks` once the target
|
||||||
|
exists)
|
||||||
|
|
||||||
Accepted divergences (no action):
|
Accepted divergences (no action):
|
||||||
|
|
||||||
- flat single-package layout with `.go` files in the repo root — fine for a
|
- flat single-package layout with `.go` files in the repo root — fine
|
||||||
small single-binary tool per the Go styleguide; the tracker audit agrees
|
for a small single-binary tool per the Go styleguide; the tracker
|
||||||
|
audit agrees
|
||||||
|
- `go test` runs without `-race` — the repo mandates `CGO_ENABLED=0`
|
||||||
|
(pure-Go builds) and the race detector requires cgo
|
||||||
|
|||||||
+45
-389
@@ -4,35 +4,20 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"database/sql"
|
"database/sql"
|
||||||
"errors"
|
"errors"
|
||||||
"fmt"
|
|
||||||
"os"
|
"os"
|
||||||
"os/signal"
|
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"slices"
|
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
|
||||||
"sync"
|
"sync"
|
||||||
"sync/atomic"
|
"sync/atomic"
|
||||||
"syscall"
|
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
"time"
|
||||||
)
|
)
|
||||||
|
|
||||||
// This file gathers the tests for scan cancellation and worker-pool
|
// poolUnwind bounds how long a goroutine is given to leave a pool
|
||||||
// unwinding. Everything it exercises lives in scan.go, so by the repo's
|
// after its context is cancelled. Only a failing run ever waits this
|
||||||
// convention of one test file per source file it would belong in
|
// long: a pool that ignored its cancellation parks forever, and this
|
||||||
// scan_test.go. It is kept separate on purpose: cancellation behaviour
|
// is what turns that into a failed assertion instead of a suite that
|
||||||
// cuts across both the walk pool and the hash pool as a single concern,
|
// hangs until the test binary's own timeout.
|
||||||
// and scan_test.go is already over 1,600 lines. That is the deliberate
|
|
||||||
// exception the convention otherwise expects to be stated.
|
|
||||||
|
|
||||||
// poolUnwind bounds how long a test waits for a cancellation to take
|
|
||||||
// effect: for a goroutine to return or a channel to close once its
|
|
||||||
// context is cancelled, or for a signal to cancel the scan's context.
|
|
||||||
// Only a failing run waits this long, and the bound is what makes that
|
|
||||||
// failure an assertion instead of a hang. A call made without it, as
|
|
||||||
// most of this file's scans are, has no bound: a regression that parks
|
|
||||||
// it is caught only as the test binary's own timeout.
|
|
||||||
const poolUnwind = 2 * time.Second
|
const poolUnwind = 2 * time.Second
|
||||||
|
|
||||||
// walkClock is a context whose cancellation is driven by the scan's
|
// walkClock is a context whose cancellation is driven by the scan's
|
||||||
@@ -43,14 +28,9 @@ const poolUnwind = 2 * time.Second
|
|||||||
//
|
//
|
||||||
// The accounting behind the n chosen by each test: every blocking
|
// The accounting behind the n chosen by each test: every blocking
|
||||||
// channel operation in the walk selects on Done, so the walk spends
|
// channel operation in the walk selects on Done, so the walk spends
|
||||||
// one consultation per file event plus a couple per directory. The
|
// one consultation per file event plus a couple per directory, while
|
||||||
// index load that runs ahead of it also consults Done, but a bounded
|
// the index load that runs ahead of it spends a small fixed number
|
||||||
// number of times that does not grow with the record count. The tests
|
// (three) whatever the record count.
|
||||||
// depend on that property, not on the bound's exact value: each test
|
|
||||||
// sets n from the consultations of the walk, plus those of the hash
|
|
||||||
// phase when it cancels mid-hash, far from both ends of the phase it
|
|
||||||
// interrupts, so the cancellation lands inside that phase whatever the
|
|
||||||
// record count.
|
|
||||||
type walkClock struct {
|
type walkClock struct {
|
||||||
n int64
|
n int64
|
||||||
seen atomic.Int64
|
seen atomic.Int64
|
||||||
@@ -110,10 +90,7 @@ const (
|
|||||||
walkCancelFilesPerDir = 20
|
walkCancelFilesPerDir = 20
|
||||||
walkCancelFiles = walkCancelDirs * walkCancelFilesPerDir
|
walkCancelFiles = walkCancelDirs * walkCancelFilesPerDir
|
||||||
walkCancelWorkers = 4
|
walkCancelWorkers = 4
|
||||||
// The most files the walkCancelWorkers directories already in
|
walkCancelInFlightDirs = walkCancelWorkers * walkCancelFilesPerDir
|
||||||
// flight when the scan is cancelled can still emit, at
|
|
||||||
// walkCancelFilesPerDir each. A file count, not a directory count.
|
|
||||||
walkCancelInFlightFiles = walkCancelWorkers * walkCancelFilesPerDir
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// walkCancelAtDone is the consultation on which the fixture's context
|
// walkCancelAtDone is the consultation on which the fixture's context
|
||||||
@@ -173,13 +150,9 @@ func assertRecordsIntact(t *testing.T, db *sql.DB, before []string) {
|
|||||||
// Every one of those records would look vanished to the update phase.
|
// Every one of those records would look vanished to the update phase.
|
||||||
// The guard is what stops the scan there, and this test is what
|
// The guard is what stops the scan there, and this test is what
|
||||||
// notices if it stops doing so: deleting the guard, or making it
|
// notices if it stops doing so: deleting the guard, or making it
|
||||||
// unreachable, makes the scan carry its truncated view into the update
|
// unreachable, makes the scan carry its truncated view into a later
|
||||||
// phase, which counts every record the walk never reached for removal.
|
// phase and fail there instead, with a wrapped error rather than the
|
||||||
//
|
// bare cancellation.
|
||||||
// The syncScan call here is not bounded by poolUnwind: a regression
|
|
||||||
// that left a worker pool parked would hang it, and that regression is
|
|
||||||
// caught only by the test binary's own timeout, not by a quick
|
|
||||||
// assertion.
|
|
||||||
//
|
//
|
||||||
//nolint:paralleltest // counts goroutines: must not run beside others
|
//nolint:paralleltest // counts goroutines: must not run beside others
|
||||||
func TestSyncScanCancelledMidWalkKeepsRecords(t *testing.T) {
|
func TestSyncScanCancelledMidWalkKeepsRecords(t *testing.T) {
|
||||||
@@ -209,10 +182,10 @@ func TestSyncScanCancelledMidWalkKeepsRecords(t *testing.T) {
|
|||||||
|
|
||||||
// assertWalkGuardAborted checks that the scan stopped at the post-walk
|
// assertWalkGuardAborted checks that the scan stopped at the post-walk
|
||||||
// guard: with a census that is neither empty (the walk really ran)
|
// guard: with a census that is neither empty (the walk really ran)
|
||||||
// nor complete (it really was cut short), and with no record counted
|
// nor complete (it really was cut short), and with the guard's own
|
||||||
// for removal. A removal count means the partial census was carried
|
// bare cancellation as the error. A wrapped error means the partial
|
||||||
// past the guard into the update phase, which is the failure this test
|
// census was carried past the guard into the hash or update phase,
|
||||||
// exists to catch.
|
// which is the failure this test exists to catch.
|
||||||
func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
|
func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
@@ -221,6 +194,12 @@ func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
|
|||||||
err, context.Canceled)
|
err, context.Canceled)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if errors.Unwrap(err) != nil {
|
||||||
|
t.Errorf("syncScan reported %q, want the guard's bare "+
|
||||||
|
"cancellation: a wrapped error means the truncated census "+
|
||||||
|
"reached a later phase", err)
|
||||||
|
}
|
||||||
|
|
||||||
if st.unchanged == 0 {
|
if st.unchanged == 0 {
|
||||||
t.Fatalf("stats = %+v: the census is empty, so the walk never "+
|
t.Fatalf("stats = %+v: the census is empty, so the walk never "+
|
||||||
"ran and the guard was reached for the wrong reason", st)
|
"ran and the guard was reached for the wrong reason", st)
|
||||||
@@ -232,11 +211,10 @@ func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// The workers drop every directory still queued once the scan is
|
// The workers drop every directory still queued once the scan is
|
||||||
// cancelled, so only the files in the directories already in flight
|
// cancelled, so only the directories already in flight can add to
|
||||||
// can add to the census after the fact. A census beyond that bound
|
// the census after the fact. A census beyond that bound would mean
|
||||||
// would mean the cancellation was not observed where it should have
|
// the cancellation was not observed where it should have been.
|
||||||
// been.
|
limit := walkCancelAtDone + walkCancelInFlightDirs
|
||||||
limit := walkCancelAtDone + walkCancelInFlightFiles
|
|
||||||
if st.unchanged > limit {
|
if st.unchanged > limit {
|
||||||
t.Errorf("census covers %d files, want at most %d: the walk kept "+
|
t.Errorf("census covers %d files, want at most %d: the walk kept "+
|
||||||
"taking directories off the queue after cancellation",
|
"taking directories off the queue after cancellation",
|
||||||
@@ -282,252 +260,6 @@ func TestSyncScanCancelledBeforeLoadIndex(t *testing.T) {
|
|||||||
assertRecordsIntact(t, db, before)
|
assertRecordsIntact(t, db, before)
|
||||||
}
|
}
|
||||||
|
|
||||||
// hashCancelAtDone is the consultation on which the mid-hash test's
|
|
||||||
// context cancels itself. The walk of buildWalkCancelTree spends about
|
|
||||||
// one per file and three per directory, and the hash phase then one per
|
|
||||||
// file hashed, so this lands about half way through the hash phase.
|
|
||||||
const hashCancelAtDone = walkCancelFiles + 3*walkCancelDirs +
|
|
||||||
walkCancelFiles/2
|
|
||||||
|
|
||||||
// TestSyncScanCancelledMidHashKeepsHashedRecords cancels a first scan
|
|
||||||
// part-way through its hash phase. The fixture holds fewer files than a
|
|
||||||
// batch, so every file hashed is still waiting to be committed: the scan
|
|
||||||
// must commit them all before it returns, and the next scan must hash
|
|
||||||
// only the rest.
|
|
||||||
func TestSyncScanCancelledMidHashKeepsHashedRecords(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
dir := buildWalkCancelTree(t)
|
|
||||||
db := openTestDB(t)
|
|
||||||
|
|
||||||
st, err := syncScan(newWalkClock(hashCancelAtDone), db,
|
|
||||||
[]string{dir}, walkCancelWorkers, false)
|
|
||||||
if !errors.Is(err, context.Canceled) {
|
|
||||||
t.Fatalf("syncScan cancelled mid-hash = %v, want %v",
|
|
||||||
err, context.Canceled)
|
|
||||||
}
|
|
||||||
|
|
||||||
if st.walked != walkCancelFiles || st.added == 0 ||
|
|
||||||
st.added >= walkCancelFiles {
|
|
||||||
t.Fatalf("stats = %+v: want the walk complete and the hash phase "+
|
|
||||||
"cut short", st)
|
|
||||||
}
|
|
||||||
|
|
||||||
if got := len(dbRecords(t, db)); got != st.added {
|
|
||||||
t.Errorf("%d records after the cancelled scan, want the %d it hashed",
|
|
||||||
got, st.added)
|
|
||||||
}
|
|
||||||
|
|
||||||
hashed := st.added
|
|
||||||
|
|
||||||
st = syncTree(t, db, dir)
|
|
||||||
if st.added != walkCancelFiles-hashed || st.unchanged != hashed {
|
|
||||||
t.Errorf("next scan stats = %+v, want %d added %d unchanged",
|
|
||||||
st, walkCancelFiles-hashed, hashed)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// storedPaths opens the database at path as report does, which fails
|
|
||||||
// unless it is a valid database, and returns its records' paths.
|
|
||||||
func storedPaths(t *testing.T, path string) []string {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
db, err := openReportDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = db.Close() }()
|
|
||||||
|
|
||||||
return recordPaths(dbRecords(t, db))
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestRunScanInterrupted calls the scan entrypoint with a context that
|
|
||||||
// is already cancelled, as when a signal arrives at once. It must return
|
|
||||||
// errInterrupted promptly with its one line on stderr and nothing on
|
|
||||||
// stdout, leave the database valid and as it was, and leave nothing in
|
|
||||||
// the way of the next scan, which must bring the database up to date.
|
|
||||||
func TestRunScanInterrupted(t *testing.T) {
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
stdout := captureStdout(t)
|
|
||||||
stderr := captureStderr(t)
|
|
||||||
dir := buildSmokeTree(t)
|
|
||||||
|
|
||||||
err := runScan(t.Context(), []string{dir}, walkCancelWorkers, false)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
before := storedPaths(t, path)
|
|
||||||
|
|
||||||
// A vanished file and a new one: the interrupted scan records
|
|
||||||
// neither.
|
|
||||||
gone := filepath.Join(dir, "a", "unique.bin")
|
|
||||||
|
|
||||||
err = os.Remove(gone)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
added := writeFile(t, dir, "a/new.bin", pattern(50, 10))
|
|
||||||
shown := len(stderr())
|
|
||||||
done := make(chan struct{})
|
|
||||||
|
|
||||||
go func() {
|
|
||||||
defer close(done)
|
|
||||||
|
|
||||||
err = runScan(cancelledContext(t), []string{dir}, walkCancelWorkers,
|
|
||||||
false)
|
|
||||||
}()
|
|
||||||
|
|
||||||
awaitReturn(t, done, "runScan")
|
|
||||||
|
|
||||||
if !errors.Is(err, errInterrupted) {
|
|
||||||
t.Fatalf("runScan on a cancelled context = %v, want %v",
|
|
||||||
err, errInterrupted)
|
|
||||||
}
|
|
||||||
|
|
||||||
want := "scan: interrupted after 0 files\n"
|
|
||||||
if got := stderr()[shown:]; got != want {
|
|
||||||
t.Errorf("stderr = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
if got := stdout(); got != "" {
|
|
||||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
|
||||||
}
|
|
||||||
|
|
||||||
assertNoSidecars(t, path)
|
|
||||||
|
|
||||||
if got := storedPaths(t, path); !slices.Equal(got, before) {
|
|
||||||
t.Errorf("records = %q after the interrupted scan, want %q",
|
|
||||||
got, before)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = runScan(t.Context(), []string{dir}, walkCancelWorkers, false)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
got := storedPaths(t, path)
|
|
||||||
if slices.Contains(got, gone) || !slices.Contains(got, added) {
|
|
||||||
t.Errorf("records = %q after the next scan, want %q gone and %q "+
|
|
||||||
"added", got, gone, added)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestRunScanInterruptedMidHash interrupts the scan entrypoint part-way
|
|
||||||
// through its hash phase, after the database is open. It must return
|
|
||||||
// errInterrupted, release the lock, end stderr with its line counting
|
|
||||||
// every file the walk reached, write nothing to stdout, close the
|
|
||||||
// database out of WAL mode, and keep the records it hashed.
|
|
||||||
func TestRunScanInterruptedMidHash(t *testing.T) {
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
stdout := captureStdout(t)
|
|
||||||
stderr := captureStderr(t)
|
|
||||||
dir := buildWalkCancelTree(t)
|
|
||||||
|
|
||||||
err := runScan(newWalkClock(hashCancelAtDone), []string{dir},
|
|
||||||
walkCancelWorkers, false)
|
|
||||||
if !errors.Is(err, errInterrupted) {
|
|
||||||
t.Fatalf("runScan interrupted mid-hash = %v, want %v",
|
|
||||||
err, errInterrupted)
|
|
||||||
}
|
|
||||||
|
|
||||||
holdScanLock(t, path)
|
|
||||||
|
|
||||||
want := fmt.Sprintf("scan: interrupted after %d files\n", walkCancelFiles)
|
|
||||||
if got := stderr(); !strings.HasSuffix(got, want) {
|
|
||||||
t.Errorf("stderr = %q, want it to end with %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
if got := stdout(); got != "" {
|
|
||||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
|
||||||
}
|
|
||||||
|
|
||||||
assertNoSidecars(t, path)
|
|
||||||
|
|
||||||
db, err := openReportDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = db.Close() }()
|
|
||||||
|
|
||||||
// A plain close also removes the sidecars, but leaves WAL mode on.
|
|
||||||
var mode string
|
|
||||||
|
|
||||||
err = db.QueryRowContext(t.Context(), "PRAGMA journal_mode").Scan(&mode)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
if mode != "delete" {
|
|
||||||
t.Errorf("journal mode = %q after the interrupted scan, want %q",
|
|
||||||
mode, "delete")
|
|
||||||
}
|
|
||||||
|
|
||||||
kept := len(dbRecords(t, db))
|
|
||||||
if kept == 0 || kept >= walkCancelFiles {
|
|
||||||
t.Errorf("%d records after the interrupted scan, want those it "+
|
|
||||||
"hashed: some but not all of the %d files", kept, walkCancelFiles)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestInterruptContextCatchesSIGTERM sends SIGTERM to the test process
|
|
||||||
// while the scan's handler is installed, and checks that it cancels the
|
|
||||||
// scan's context.
|
|
||||||
//
|
|
||||||
//nolint:paralleltest // signals the whole process: must not run beside a scan
|
|
||||||
func TestInterruptContextCatchesSIGTERM(t *testing.T) {
|
|
||||||
// Caught here as well, so that a handler that misses SIGTERM fails
|
|
||||||
// this test instead of ending the test process.
|
|
||||||
caught := make(chan os.Signal, 1)
|
|
||||||
signal.Notify(caught, syscall.SIGTERM)
|
|
||||||
|
|
||||||
defer signal.Stop(caught)
|
|
||||||
|
|
||||||
ctx, stop := interruptContext(t.Context())
|
|
||||||
defer stop()
|
|
||||||
|
|
||||||
err := syscall.Kill(os.Getpid(), syscall.SIGTERM)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
select {
|
|
||||||
case <-ctx.Done():
|
|
||||||
case <-time.After(poolUnwind):
|
|
||||||
t.Fatal("SIGTERM did not cancel the scan's context")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestCommitFullBatchKeepsFailedBatch checks that a full batch whose
|
|
||||||
// commit fails, as it does once the scan is interrupted, stays in the
|
|
||||||
// batch, so that syncScan's final commit saves it.
|
|
||||||
func TestCommitFullBatchKeepsFailedBatch(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
s := &scanState{db: openTestDB(t)}
|
|
||||||
for i := range updateBatchSize {
|
|
||||||
s.batch = append(s.batch, scanRec{path: "/f" + strconv.Itoa(i)})
|
|
||||||
}
|
|
||||||
|
|
||||||
err := s.commitFullBatch(cancelledContext(t))
|
|
||||||
if !errors.Is(err, context.Canceled) {
|
|
||||||
t.Fatalf("commitFullBatch on a cancelled context = %v, want %v",
|
|
||||||
err, context.Canceled)
|
|
||||||
}
|
|
||||||
|
|
||||||
if len(s.batch) != updateBatchSize {
|
|
||||||
t.Errorf("batch holds %d records after the failed commit, want %d",
|
|
||||||
len(s.batch), updateBatchSize)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// drainClosed counts the values received from ch until it closes,
|
// drainClosed counts the values received from ch until it closes,
|
||||||
// failing the test if it does not close within poolUnwind. A pool that
|
// failing the test if it does not close within poolUnwind. A pool that
|
||||||
// ignored its cancellation leaves its channel open with its goroutines
|
// ignored its cancellation leaves its channel open with its goroutines
|
||||||
@@ -596,49 +328,20 @@ func TestSendEventAbandonsBlockedSend(t *testing.T) {
|
|||||||
awaitReturn(t, done, "sendEvent")
|
awaitReturn(t, done, "sendEvent")
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestWalkOneDirStopsWhenCancelled checks that a cancelled scan stops
|
|
||||||
// reading a directory instead of going through the rest of its
|
|
||||||
// entries. A walk that kept going would return the subdirectory below
|
|
||||||
// to descend into. Unlike a file event, that return is not a send the
|
|
||||||
// cancellation can abandon, so the test catches the regression every
|
|
||||||
// time.
|
|
||||||
func TestWalkOneDirStopsWhenCancelled(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
dir := t.TempDir()
|
|
||||||
|
|
||||||
err := os.Mkdir(filepath.Join(dir, "sub"), 0o750)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Unbuffered and unread: on a cancelled scan every send gives up.
|
|
||||||
events := make(chan walkEvent)
|
|
||||||
|
|
||||||
subs := walkOneDir(cancelledContext(t), dirJob{path: dir}, false, events)
|
|
||||||
if len(subs) != 0 {
|
|
||||||
t.Errorf("cancelled walkOneDir returned %+v to descend into, "+
|
|
||||||
"want none", subs)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestWalkWorkersDropQueuedDirs checks that cancelled walk workers keep
|
// TestWalkWorkersDropQueuedDirs checks that cancelled walk workers keep
|
||||||
// reading jobs and drop the directories rather than stopping their
|
// reading jobs and drop the directories rather than stopping their
|
||||||
// read: the range over jobs has to run out for the pool to tear down
|
// read: the range over jobs has to run out for the pool to tear down
|
||||||
// and close its event stream. The queued directory does not exist, so
|
// and close its event stream.
|
||||||
// a worker that walked it anyway would send a warning before
|
|
||||||
// walkOneDir's own cancellation check could stop it. On a cancelled
|
|
||||||
// scan that send delivers or gives up at random, so with 64 jobs
|
|
||||||
// queued the regression has a one in 2^64 chance of passing.
|
|
||||||
func TestWalkWorkersDropQueuedDirs(t *testing.T) {
|
func TestWalkWorkersDropQueuedDirs(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
missing := filepath.Join(t.TempDir(), "missing")
|
dir := t.TempDir()
|
||||||
|
writeEmptyFiles(t, dir, walkCancelFilesPerDir)
|
||||||
|
|
||||||
jobs, _, events := startWalkWorkers(cancelledContext(t), 2, false)
|
jobs, _, events := startWalkWorkers(cancelledContext(t), 2, false)
|
||||||
|
|
||||||
for range 64 {
|
for range 4 {
|
||||||
jobs <- dirJob{path: missing}
|
jobs <- dirJob{path: dir}
|
||||||
}
|
}
|
||||||
|
|
||||||
close(jobs)
|
close(jobs)
|
||||||
@@ -709,11 +412,7 @@ func TestDispatchDirsClosesJobsWhenCancelled(t *testing.T) {
|
|||||||
|
|
||||||
// TestFeedHashJobsClosesJobsWhenCancelled checks that the hash feeder
|
// TestFeedHashJobsClosesJobsWhenCancelled checks that the hash feeder
|
||||||
// abandons the runs it has not queued yet and still closes the job
|
// abandons the runs it has not queued yet and still closes the job
|
||||||
// channel, which is what lets the workers' range terminate. The
|
// channel, which is what lets the workers' range terminate.
|
||||||
// receive on jobs below is not bounded: a feeder that returned without
|
|
||||||
// closing jobs would leave that receive with no sender and no close, so
|
|
||||||
// this regression is caught by the test binary's timeout rather than by
|
|
||||||
// a bounded assertion.
|
|
||||||
func TestFeedHashJobsClosesJobsWhenCancelled(t *testing.T) {
|
func TestFeedHashJobsClosesJobsWhenCancelled(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
@@ -739,84 +438,41 @@ func TestFeedHashJobsClosesJobsWhenCancelled(t *testing.T) {
|
|||||||
// TestHashWorkerDropsQueuedRuns checks that a cancelled hash worker
|
// TestHashWorkerDropsQueuedRuns checks that a cancelled hash worker
|
||||||
// keeps reading jobs and drops the runs rather than reading files
|
// keeps reading jobs and drops the runs rather than reading files
|
||||||
// nobody wants the hashes of — while still letting the range run out
|
// nobody wants the hashes of — while still letting the range run out
|
||||||
// so the pool tears down. The hash function records that it was
|
// so the pool tears down. The queued run names a file that does not
|
||||||
// called, so a worker that hashed the queued run anyway is caught
|
// exist, so a worker that hashed it anyway would produce a result.
|
||||||
// every time.
|
|
||||||
func TestHashWorkerDropsQueuedRuns(t *testing.T) {
|
func TestHashWorkerDropsQueuedRuns(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
done := make(chan struct{})
|
done := make(chan struct{})
|
||||||
jobs := make(chan []fileRec, 1)
|
jobs := make(chan []fileRec, 1)
|
||||||
results := make(chan hashResult)
|
results := make(chan hashResult, 1)
|
||||||
|
|
||||||
jobs <- []fileRec{{path: filepath.Join(t.TempDir(), "missing"), size: 1}}
|
run := []fileRec{{path: filepath.Join(t.TempDir(), "missing"), size: 1}}
|
||||||
|
|
||||||
|
jobs <- run
|
||||||
|
|
||||||
close(jobs)
|
close(jobs)
|
||||||
|
|
||||||
var hashed atomic.Bool
|
|
||||||
|
|
||||||
hash := func(path string, size int64) (string, string, string, error) {
|
|
||||||
hashed.Store(true)
|
|
||||||
|
|
||||||
return hashSignature(path, size)
|
|
||||||
}
|
|
||||||
|
|
||||||
go func() {
|
go func() {
|
||||||
defer close(done)
|
defer close(done)
|
||||||
|
|
||||||
hashWorker(cancelledContext(t), jobs, results, hash)
|
hashWorker(cancelledContext(t), jobs, results, hashSignature)
|
||||||
}()
|
}()
|
||||||
|
|
||||||
awaitReturn(t, done, "hashWorker")
|
awaitReturn(t, done, "hashWorker")
|
||||||
|
|
||||||
if hashed.Load() {
|
select {
|
||||||
t.Error("cancelled hash worker hashed the queued run, want it dropped")
|
case r := <-results:
|
||||||
|
t.Errorf("cancelled hash worker produced %+v, want the run dropped",
|
||||||
|
r)
|
||||||
|
default:
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestHashWorkerAbandonsBlockedSend checks that a hash worker with a
|
|
||||||
// result to deliver and nobody to deliver it to leaves once the scan
|
|
||||||
// is cancelled, instead of holding the pool open. The scan tests do
|
|
||||||
// not catch this: stop drains results, which frees a parked worker
|
|
||||||
// anyway.
|
|
||||||
func TestHashWorkerAbandonsBlockedSend(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
ctx, cancel := context.WithCancel(t.Context())
|
|
||||||
defer cancel()
|
|
||||||
|
|
||||||
done := make(chan struct{})
|
|
||||||
jobs := make(chan []fileRec, 1)
|
|
||||||
// Unbuffered and unread, with jobs left open: the worker's only way
|
|
||||||
// out is the cancellation case beside its send.
|
|
||||||
results := make(chan hashResult)
|
|
||||||
|
|
||||||
jobs <- []fileRec{{path: filepath.Join(t.TempDir(), "missing"), size: 1}}
|
|
||||||
|
|
||||||
// The scan is cancelled while the worker hashes, so the worker has
|
|
||||||
// already passed the check that drops queued runs.
|
|
||||||
hash := func(path string, size int64) (string, string, string, error) {
|
|
||||||
cancel()
|
|
||||||
|
|
||||||
return hashSignature(path, size)
|
|
||||||
}
|
|
||||||
|
|
||||||
go func() {
|
|
||||||
defer close(done)
|
|
||||||
|
|
||||||
hashWorker(ctx, jobs, results, hash)
|
|
||||||
}()
|
|
||||||
|
|
||||||
awaitReturn(t, done, "hashWorker")
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestHashPhaseCancelledReturnsContextError checks the result loop's
|
// TestHashPhaseCancelledReturnsContextError checks the result loop's
|
||||||
// own exit: with the pool cancelled, no result will ever arrive, and
|
// own exit: with the pool cancelled, no result will ever arrive, and
|
||||||
// the loop must leave through the cancellation rather than wait for a
|
// the loop must leave through the cancellation rather than wait for a
|
||||||
// receive that cannot happen. This call is not bounded by poolUnwind: a
|
// receive that cannot happen.
|
||||||
// loop that dropped its cancellation case would block on that receive,
|
|
||||||
// so the regression surfaces as the test binary's timeout rather than
|
|
||||||
// as a bounded assertion.
|
|
||||||
func TestHashPhaseCancelledReturnsContextError(t *testing.T) {
|
func TestHashPhaseCancelledReturnsContextError(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
|
|||||||
@@ -6,14 +6,11 @@ import (
|
|||||||
"errors"
|
"errors"
|
||||||
"fmt"
|
"fmt"
|
||||||
"io/fs"
|
"io/fs"
|
||||||
"net/url"
|
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"slices"
|
"slices"
|
||||||
"strconv"
|
"strconv"
|
||||||
"time"
|
|
||||||
|
|
||||||
"golang.org/x/sys/unix"
|
|
||||||
// The pure-Go SQLite driver, registered as "sqlite"; keeps cgo
|
// The pure-Go SQLite driver, registered as "sqlite"; keeps cgo
|
||||||
// disabled.
|
// disabled.
|
||||||
_ "modernc.org/sqlite"
|
_ "modernc.org/sqlite"
|
||||||
@@ -35,41 +32,26 @@ const schemaVersion = 1
|
|||||||
// scan.
|
// scan.
|
||||||
const dbDirPerm = 0o755
|
const dbDirPerm = 0o755
|
||||||
|
|
||||||
// lockFilePerm is the mode for the scan lock file. Anyone who can open
|
|
||||||
// the file can hold the lock and keep every scan from running, so it
|
|
||||||
// is open to its owner only.
|
|
||||||
const lockFilePerm = 0o600
|
|
||||||
|
|
||||||
// createTableSQL is the schema applied to a fresh database. Paths are
|
// createTableSQL is the schema applied to a fresh database. Paths are
|
||||||
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8. mtime
|
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8.
|
||||||
// holds whole Unix seconds and mtime_nsec the nanoseconds within that
|
|
||||||
// second.
|
|
||||||
const createTableSQL = `
|
const createTableSQL = `
|
||||||
CREATE TABLE files (
|
CREATE TABLE files (
|
||||||
path BLOB PRIMARY KEY,
|
path BLOB PRIMARY KEY,
|
||||||
size INTEGER NOT NULL,
|
size INTEGER NOT NULL,
|
||||||
mtime INTEGER NOT NULL,
|
mtime INTEGER NOT NULL,
|
||||||
mtime_nsec INTEGER NOT NULL,
|
|
||||||
head TEXT NOT NULL,
|
head TEXT NOT NULL,
|
||||||
tail TEXT NOT NULL,
|
tail TEXT NOT NULL,
|
||||||
content TEXT NOT NULL
|
content TEXT NOT NULL
|
||||||
) WITHOUT ROWID
|
) WITHOUT ROWID
|
||||||
`
|
`
|
||||||
|
|
||||||
// createIndexSQL indexes the records by signature, so report can have
|
|
||||||
// SQLite group them without sorting the whole table.
|
|
||||||
const createIndexSQL = `
|
|
||||||
CREATE INDEX files_signature ON files (size, head, tail, content)
|
|
||||||
`
|
|
||||||
|
|
||||||
// upsertSQL inserts one file record, replacing any existing record for
|
// upsertSQL inserts one file record, replacing any existing record for
|
||||||
// the same path.
|
// the same path.
|
||||||
const upsertSQL = `
|
const upsertSQL = `
|
||||||
INSERT INTO files (path, size, mtime, mtime_nsec, head, tail, content)
|
INSERT INTO files (path, size, mtime, head, tail, content)
|
||||||
VALUES (?, ?, ?, ?, ?, ?, ?)
|
VALUES (?, ?, ?, ?, ?, ?)
|
||||||
ON CONFLICT (path) DO UPDATE SET
|
ON CONFLICT (path) DO UPDATE SET
|
||||||
size = excluded.size, mtime = excluded.mtime,
|
size = excluded.size, mtime = excluded.mtime,
|
||||||
mtime_nsec = excluded.mtime_nsec,
|
|
||||||
head = excluded.head, tail = excluded.tail,
|
head = excluded.head, tail = excluded.tail,
|
||||||
content = excluded.content
|
content = excluded.content
|
||||||
`
|
`
|
||||||
@@ -82,10 +64,6 @@ var errNoDatabase = errors.New(
|
|||||||
// does not understand.
|
// does not understand.
|
||||||
var errSchemaVersion = errors.New("unsupported database schema version")
|
var errSchemaVersion = errors.New("unsupported database schema version")
|
||||||
|
|
||||||
// errScanRunning reports that another scan holds the lock on the
|
|
||||||
// database.
|
|
||||||
var errScanRunning = errors.New("another scan is running")
|
|
||||||
|
|
||||||
// databasePath resolves the database location: SFDUPES_DATABASE when
|
// databasePath resolves the database location: SFDUPES_DATABASE when
|
||||||
// set and non-empty, the compiled-in default otherwise.
|
// set and non-empty, the compiled-in default otherwise.
|
||||||
func databasePath() string {
|
func databasePath() string {
|
||||||
@@ -96,36 +74,16 @@ func databasePath() string {
|
|||||||
return defaultDatabasePath
|
return defaultDatabasePath
|
||||||
}
|
}
|
||||||
|
|
||||||
// scanParams are the connection parameters for scan: read-write, with
|
// openDB opens the SQLite database at path with WAL journaling and a
|
||||||
// WAL journaling and a busy timeout, so a report can run while a cron
|
// busy timeout, so a report can run while a cron scan is in progress.
|
||||||
// scan is in progress. closeScanDatabase leaves WAL mode again.
|
// It does not create or verify the schema.
|
||||||
const scanParams = "_pragma=busy_timeout(10000)" +
|
func openDB(path string) (*sql.DB, error) {
|
||||||
|
dsn := "file:" + path +
|
||||||
|
"?_pragma=busy_timeout(10000)" +
|
||||||
"&_pragma=journal_mode(WAL)" +
|
"&_pragma=journal_mode(WAL)" +
|
||||||
"&_pragma=synchronous(NORMAL)"
|
"&_pragma=synchronous(NORMAL)"
|
||||||
|
|
||||||
// reportParams are the connection parameters for report and trees:
|
db, err := sql.Open("sqlite", dsn)
|
||||||
// read-only, with the same busy timeout. They set no journal mode,
|
|
||||||
// because setting one is a write.
|
|
||||||
const reportParams = "mode=ro" +
|
|
||||||
"&_pragma=busy_timeout(10000)" +
|
|
||||||
"&_pragma=query_only(1)"
|
|
||||||
|
|
||||||
// openDB opens the SQLite database at path with the connection
|
|
||||||
// parameters params. It does not create or verify the schema.
|
|
||||||
func openDB(path, params string) (*sql.DB, error) {
|
|
||||||
// The path is escaped into a file: URI, so ?, # and % in it stay
|
|
||||||
// part of the file name. SQLite reads what follows file:// up to
|
|
||||||
// the next / as a host name, so an absolute path goes after an
|
|
||||||
// empty host (file:///abs) and a relative path goes without one
|
|
||||||
// (file:rel).
|
|
||||||
uri := url.URL{
|
|
||||||
Scheme: "file",
|
|
||||||
OmitHost: !filepath.IsAbs(path),
|
|
||||||
Path: path,
|
|
||||||
RawQuery: params,
|
|
||||||
}
|
|
||||||
|
|
||||||
db, err := sql.Open("sqlite", uri.String())
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("open database %s: %w", path, err)
|
return nil, fmt.Errorf("open database %s: %w", path, err)
|
||||||
}
|
}
|
||||||
@@ -138,43 +96,6 @@ func openDB(path, params string) (*sql.DB, error) {
|
|||||||
return db, nil
|
return db, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// lockScanDatabase takes the lock that keeps a second scan off the
|
|
||||||
// database at path: an exclusive flock(2) on the file beside it named
|
|
||||||
// path with ".lock" appended, created along with the database's parent
|
|
||||||
// directory if missing. A lock held by another scan fails at once
|
|
||||||
// instead of waiting. The lock lasts until the returned file is closed
|
|
||||||
// or the process ends. The file is never deleted: a scan that deleted
|
|
||||||
// it would let the next scan lock a new file while another still holds
|
|
||||||
// the old one.
|
|
||||||
func lockScanDatabase(path string) (*os.File, error) {
|
|
||||||
err := os.MkdirAll(filepath.Dir(path), dbDirPerm)
|
|
||||||
if err != nil {
|
|
||||||
return nil, fmt.Errorf("create database directory: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
lockPath := path + ".lock"
|
|
||||||
|
|
||||||
//nolint:gosec // the operator chooses the database path
|
|
||||||
f, err := os.OpenFile(lockPath, os.O_RDWR|os.O_CREATE, lockFilePerm)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
|
|
||||||
err = unix.Flock(int(f.Fd()), unix.LOCK_EX|unix.LOCK_NB)
|
|
||||||
if err != nil {
|
|
||||||
_ = f.Close()
|
|
||||||
|
|
||||||
if errors.Is(err, unix.EWOULDBLOCK) {
|
|
||||||
return nil, fmt.Errorf("%w (lock held on %s)",
|
|
||||||
errScanRunning, lockPath)
|
|
||||||
}
|
|
||||||
|
|
||||||
return nil, fmt.Errorf("lock %s: %w", lockPath, err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return f, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// openScanDatabase opens the database for the scan subcommand, creating
|
// openScanDatabase opens the database for the scan subcommand, creating
|
||||||
// the file, its parent directory, and the schema as needed.
|
// the file, its parent directory, and the schema as needed.
|
||||||
func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
|
func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
|
||||||
@@ -183,7 +104,7 @@ func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
|
|||||||
return nil, fmt.Errorf("create database directory: %w", err)
|
return nil, fmt.Errorf("create database directory: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
db, err := openDB(path, scanParams)
|
db, err := openDB(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
@@ -198,24 +119,6 @@ func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
|
|||||||
return db, nil
|
return db, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// closeScanDatabase switches the database at path from WAL back to
|
|
||||||
// rollback-journal mode and closes it. Out of WAL mode the database
|
|
||||||
// file alone holds the whole database, so a reader needs no -wal or
|
|
||||||
// -shm file beside it, nor write access to create them. The switch
|
|
||||||
// fails while a report has the database open; the database then stays
|
|
||||||
// in WAL mode, still readable, until a later scan closes it.
|
|
||||||
func closeScanDatabase(ctx context.Context, db *sql.DB, path string) {
|
|
||||||
// Runs on the way out of a cancelled scan too.
|
|
||||||
_, err := db.ExecContext(context.WithoutCancel(ctx),
|
|
||||||
"PRAGMA journal_mode = DELETE")
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintf(os.Stderr, "scan: database %s left in WAL mode: %v\n",
|
|
||||||
path, err)
|
|
||||||
}
|
|
||||||
|
|
||||||
_ = db.Close()
|
|
||||||
}
|
|
||||||
|
|
||||||
// openReportDatabase opens an existing database for the report and
|
// openReportDatabase opens an existing database for the report and
|
||||||
// trees subcommands. A missing database file is an error directing the
|
// trees subcommands. A missing database file is an error directing the
|
||||||
// user to run scan first; the schema version must match exactly.
|
// user to run scan first; the schema version must match exactly.
|
||||||
@@ -231,18 +134,12 @@ func openReportDatabase(ctx context.Context,
|
|||||||
return nil, fmt.Errorf("database: %w", err)
|
return nil, fmt.Errorf("database: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
db, err := openDB(path, reportParams)
|
db, err := openDB(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
v, err := userVersion(ctx, db)
|
v, err := userVersion(ctx, db)
|
||||||
if err == nil && v == 0 {
|
|
||||||
// An empty database passes this check and fails the version
|
|
||||||
// check below.
|
|
||||||
err = checkUnversioned(ctx, db)
|
|
||||||
}
|
|
||||||
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
_ = db.Close()
|
_ = db.Close()
|
||||||
|
|
||||||
@@ -269,11 +166,6 @@ func initSchema(ctx context.Context, db *sql.DB) error {
|
|||||||
|
|
||||||
switch v {
|
switch v {
|
||||||
case 0:
|
case 0:
|
||||||
err = checkUnversioned(ctx, db)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
|
|
||||||
return createSchema(ctx, db)
|
return createSchema(ctx, db)
|
||||||
case schemaVersion:
|
case schemaVersion:
|
||||||
return nil
|
return nil
|
||||||
@@ -283,65 +175,20 @@ func initSchema(ctx context.Context, db *sql.DB) error {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// checkUnversioned checks a database at user_version 0 before it is
|
|
||||||
// taken for an empty one. createSchema creates the files table and
|
|
||||||
// sets the version together, so a files table at version 0 was made by
|
|
||||||
// something else. Adopting it could corrupt unrelated data, so that is
|
|
||||||
// a schema-version error telling the operator to remove the file and
|
|
||||||
// rescan.
|
|
||||||
func checkUnversioned(ctx context.Context, db *sql.DB) error {
|
|
||||||
var name string
|
|
||||||
|
|
||||||
err := db.QueryRowContext(ctx,
|
|
||||||
"SELECT name FROM sqlite_master "+
|
|
||||||
"WHERE type = 'table' AND name = 'files'").Scan(&name)
|
|
||||||
|
|
||||||
switch {
|
|
||||||
case err == nil:
|
|
||||||
return fmt.Errorf(
|
|
||||||
"has a files table but no schema version; "+
|
|
||||||
"remove the file and rescan: %w", errSchemaVersion)
|
|
||||||
case errors.Is(err, sql.ErrNoRows):
|
|
||||||
return nil
|
|
||||||
default:
|
|
||||||
return fmt.Errorf("check for files table: %w", err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// createSchema applies the schema to a fresh database and stamps the
|
// createSchema applies the schema to a fresh database and stamps the
|
||||||
// schema version in one transaction, so a creation stopped partway, by
|
// schema version.
|
||||||
// an interrupt or an error, leaves an empty database the next scan
|
|
||||||
// sets up, never a files table at version 0, which checkUnversioned
|
|
||||||
// refuses.
|
|
||||||
func createSchema(ctx context.Context, db *sql.DB) error {
|
func createSchema(ctx context.Context, db *sql.DB) error {
|
||||||
tx, err := db.BeginTx(ctx, nil)
|
_, err := db.ExecContext(ctx, createTableSQL)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("create schema: %w", err)
|
return fmt.Errorf("create schema: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
defer func() { _ = tx.Rollback() }()
|
_, err = db.ExecContext(ctx,
|
||||||
|
|
||||||
_, err = tx.ExecContext(ctx, createTableSQL)
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("create schema: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err = tx.ExecContext(ctx, createIndexSQL)
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("create schema: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err = tx.ExecContext(ctx,
|
|
||||||
"PRAGMA user_version = "+strconv.Itoa(schemaVersion))
|
"PRAGMA user_version = "+strconv.Itoa(schemaVersion))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("set schema version: %w", err)
|
return fmt.Errorf("set schema version: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
err = tx.Commit()
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("create schema: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -357,115 +204,40 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) {
|
|||||||
return v, nil
|
return v, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// loadFileRows streams every record to fn in path order: byte order,
|
// loadFileRows reads every record from the files table.
|
||||||
// which is the order of the primary key, so SQLite does not sort.
|
func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
||||||
func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
|
|
||||||
rows, err := db.QueryContext(ctx,
|
rows, err := db.QueryContext(ctx,
|
||||||
"SELECT path, size, mtime, mtime_nsec, head, tail, content "+
|
"SELECT path, size, mtime, head, tail, content FROM files")
|
||||||
"FROM files ORDER BY path")
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read records: %w", err)
|
return nil, fmt.Errorf("read records: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
defer func() { _ = rows.Close() }()
|
defer func() { _ = rows.Close() }()
|
||||||
|
|
||||||
|
var recs []scanRec
|
||||||
|
|
||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
var (
|
var (
|
||||||
path []byte
|
path []byte
|
||||||
sec, nsec int64
|
|
||||||
r scanRec
|
r scanRec
|
||||||
)
|
)
|
||||||
|
|
||||||
err = rows.Scan(&path, &r.size, &sec, &nsec, &r.head, &r.tail,
|
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
|
||||||
&r.content)
|
&r.content)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read record: %w", err)
|
return nil, fmt.Errorf("read record: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
r.path = string(path)
|
r.path = string(path)
|
||||||
r.mtime = time.Unix(sec, nsec)
|
recs = append(recs, r)
|
||||||
fn(r)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
err = rows.Err()
|
err = rows.Err()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read records: %w", err)
|
return nil, fmt.Errorf("read records: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
return nil
|
return recs, nil
|
||||||
}
|
|
||||||
|
|
||||||
// dupeRowsSQL selects every record in a duplicate group, with the
|
|
||||||
// group's first path. A group is the records with a content hash that
|
|
||||||
// share a size, head, tail, and content, when there are two or more of
|
|
||||||
// them. The rows come in report order: groups by size descending, then
|
|
||||||
// by first path, and each group's paths ascending.
|
|
||||||
const dupeRowsSQL = `
|
|
||||||
SELECT g.first, f.path, f.size
|
|
||||||
FROM files AS f
|
|
||||||
JOIN (
|
|
||||||
SELECT size, head, tail, content, MIN(path) AS first
|
|
||||||
FROM files
|
|
||||||
WHERE content <> ''
|
|
||||||
GROUP BY size, head, tail, content
|
|
||||||
HAVING COUNT(*) > 1
|
|
||||||
) AS g USING (size, head, tail, content)
|
|
||||||
ORDER BY f.size DESC, g.first, f.path
|
|
||||||
`
|
|
||||||
|
|
||||||
// loadDupeRows streams the rows of dupeRowsSQL to fn and returns the
|
|
||||||
// number of records in the database. The count and the rows are read
|
|
||||||
// in one transaction, so they agree while a scan is committing. An
|
|
||||||
// error from fn stops the reading and is returned as it is.
|
|
||||||
func loadDupeRows(ctx context.Context, db *sql.DB,
|
|
||||||
fn func(first, path string, size int64) error,
|
|
||||||
) (int, error) {
|
|
||||||
// Everything goes through tx: the report connection is the only
|
|
||||||
// one, so a query on db would wait for tx forever.
|
|
||||||
tx, err := db.BeginTx(ctx, &sql.TxOptions{ReadOnly: true})
|
|
||||||
if err != nil {
|
|
||||||
return 0, fmt.Errorf("read records: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = tx.Rollback() }()
|
|
||||||
|
|
||||||
var records int
|
|
||||||
|
|
||||||
err = tx.QueryRowContext(ctx, "SELECT COUNT(*) FROM files").Scan(&records)
|
|
||||||
if err != nil {
|
|
||||||
return 0, fmt.Errorf("read records: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
rows, err := tx.QueryContext(ctx, dupeRowsSQL)
|
|
||||||
if err != nil {
|
|
||||||
return 0, fmt.Errorf("read records: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = rows.Close() }()
|
|
||||||
|
|
||||||
for rows.Next() {
|
|
||||||
var (
|
|
||||||
first, path []byte
|
|
||||||
size int64
|
|
||||||
)
|
|
||||||
|
|
||||||
err = rows.Scan(&first, &path, &size)
|
|
||||||
if err != nil {
|
|
||||||
return 0, fmt.Errorf("read record: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = fn(string(first), string(path), size)
|
|
||||||
if err != nil {
|
|
||||||
return 0, err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
err = rows.Err()
|
|
||||||
if err != nil {
|
|
||||||
return 0, fmt.Errorf("read records: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return records, nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// loadFileMeta streams every record's path, size, mtime, and whether
|
// loadFileMeta streams every record's path, size, mtime, and whether
|
||||||
@@ -473,10 +245,10 @@ func loadDupeRows(ctx context.Context, db *sql.DB,
|
|||||||
// values, and skipping the hash columns keeps the scan's in-memory
|
// values, and skipping the hash columns keeps the scan's in-memory
|
||||||
// index small on multi-million-file databases.
|
// index small on multi-million-file databases.
|
||||||
func loadFileMeta(ctx context.Context, db *sql.DB,
|
func loadFileMeta(ctx context.Context, db *sql.DB,
|
||||||
fn func(path string, size int64, mtime time.Time, hashed bool),
|
fn func(path string, size, mtime int64, hashed bool),
|
||||||
) error {
|
) error {
|
||||||
rows, err := db.QueryContext(ctx,
|
rows, err := db.QueryContext(ctx,
|
||||||
"SELECT path, size, mtime, mtime_nsec, head <> '' FROM files")
|
"SELECT path, size, mtime, head <> '' FROM files")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read records: %w", err)
|
return fmt.Errorf("read records: %w", err)
|
||||||
}
|
}
|
||||||
@@ -486,16 +258,16 @@ func loadFileMeta(ctx context.Context, db *sql.DB,
|
|||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
var (
|
var (
|
||||||
path []byte
|
path []byte
|
||||||
size, sec, nsec int64
|
size, mtime int64
|
||||||
hashed int64
|
hashed int64
|
||||||
)
|
)
|
||||||
|
|
||||||
err = rows.Scan(&path, &size, &sec, &nsec, &hashed)
|
err = rows.Scan(&path, &size, &mtime, &hashed)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read record: %w", err)
|
return fmt.Errorf("read record: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn(string(path), size, time.Unix(sec, nsec), hashed != 0)
|
fn(string(path), size, mtime, hashed != 0)
|
||||||
}
|
}
|
||||||
|
|
||||||
err = rows.Err()
|
err = rows.Err()
|
||||||
@@ -514,7 +286,7 @@ func loadFileMeta(ctx context.Context, db *sql.DB,
|
|||||||
// memory; the rows come ordered by size, head, and tail, so each
|
// memory; the rows come ordered by size, head, and tail, so each
|
||||||
// group's rows arrive together.
|
// group's rows arrive together.
|
||||||
const contentCandidatesSQL = `
|
const contentCandidatesSQL = `
|
||||||
SELECT f.path, f.size, f.mtime, f.mtime_nsec, f.head, f.tail, f.content <> ''
|
SELECT f.path, f.size, f.mtime, f.head, f.tail, f.content <> ''
|
||||||
FROM files AS f
|
FROM files AS f
|
||||||
JOIN (
|
JOIN (
|
||||||
SELECT size, head, tail
|
SELECT size, head, tail
|
||||||
@@ -541,19 +313,16 @@ func loadContentCandidates(ctx context.Context, db *sql.DB,
|
|||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
var (
|
var (
|
||||||
path []byte
|
path []byte
|
||||||
sec, nsec int64
|
|
||||||
r scanRec
|
r scanRec
|
||||||
hashed int64
|
hashed int64
|
||||||
)
|
)
|
||||||
|
|
||||||
err = rows.Scan(&path, &r.size, &sec, &nsec, &r.head, &r.tail,
|
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, &hashed)
|
||||||
&hashed)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read record: %w", err)
|
return fmt.Errorf("read record: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
r.path = string(path)
|
r.path = string(path)
|
||||||
r.mtime = time.Unix(sec, nsec)
|
|
||||||
fn(r, hashed != 0)
|
fn(r, hashed != 0)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -637,8 +406,8 @@ func execUpserts(ctx context.Context, tx *sql.Tx, upserts []scanRec,
|
|||||||
defer func() { _ = st.Close() }()
|
defer func() { _ = st.Close() }()
|
||||||
|
|
||||||
for _, r := range upserts {
|
for _, r := range upserts {
|
||||||
_, err = st.ExecContext(ctx, []byte(r.path), r.size,
|
_, err = st.ExecContext(ctx,
|
||||||
r.mtime.Unix(), r.mtime.Nanosecond(), r.head, r.tail, r.content)
|
[]byte(r.path), r.size, r.mtime, r.head, r.tail, r.content)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("upsert %s: %w", r.path, err)
|
return fmt.Errorf("upsert %s: %w", r.path, err)
|
||||||
}
|
}
|
||||||
|
|||||||
+30
-142
@@ -5,12 +5,10 @@ import (
|
|||||||
"database/sql"
|
"database/sql"
|
||||||
"errors"
|
"errors"
|
||||||
"fmt"
|
"fmt"
|
||||||
"os"
|
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"slices"
|
"slices"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// testDBPath returns a database path inside a fresh temp dir.
|
// testDBPath returns a database path inside a fresh temp dir.
|
||||||
@@ -74,78 +72,9 @@ func TestOpenScanDatabaseCreates(t *testing.T) {
|
|||||||
|
|
||||||
defer func() { _ = db.Close() }()
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
if recs := dbRecords(t, db); len(recs) != 0 {
|
recs, err := loadFileRows(t.Context(), db)
|
||||||
t.Fatalf("records = %v, want none", recs)
|
if err != nil || len(recs) != 0 {
|
||||||
}
|
t.Fatalf("loadFileRows = %v, %v; want empty, nil", recs, err)
|
||||||
}
|
|
||||||
|
|
||||||
func TestOpenDatabaseUnversionedForeign(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
// A database that has a files table but user_version 0, written by
|
|
||||||
// some other tool. report, trees and scan must refuse it with the
|
|
||||||
// schema-version error, not adopt it and not emit a raw SQLite
|
|
||||||
// "table files already exists".
|
|
||||||
path := testDBPath(t)
|
|
||||||
|
|
||||||
db, err := sql.Open("sqlite", path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err = db.ExecContext(t.Context(), "CREATE TABLE files (x INTEGER)")
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
_ = db.Close()
|
|
||||||
|
|
||||||
_, err = openReportDatabase(t.Context(), path)
|
|
||||||
if !errors.Is(err, errSchemaVersion) ||
|
|
||||||
!strings.Contains(err.Error(), "remove the file and rescan") {
|
|
||||||
t.Fatalf("report: err = %v, want errSchemaVersion telling the "+
|
|
||||||
"operator to remove the file and rescan", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err = openScanDatabase(t.Context(), path)
|
|
||||||
if !errors.Is(err, errSchemaVersion) ||
|
|
||||||
!strings.Contains(err.Error(), "remove the file and rescan") {
|
|
||||||
t.Fatalf("scan: err = %v, want errSchemaVersion telling the "+
|
|
||||||
"operator to remove the file and rescan", err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestSchemaCreationStoppedPartway(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
// A first scan stopped while creating the schema must leave a
|
|
||||||
// database the next scan accepts. max_page_count(2) leaves room for
|
|
||||||
// the files table but not its index, so schema creation fails right
|
|
||||||
// after CREATE TABLE, a point an interrupt could also stop it at.
|
|
||||||
path := testDBPath(t)
|
|
||||||
|
|
||||||
db, err := openDB(path, scanParams+"&_pragma=max_page_count(2)")
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = initSchema(t.Context(), db)
|
|
||||||
_ = db.Close()
|
|
||||||
|
|
||||||
if err == nil {
|
|
||||||
t.Fatal("initSchema with no room for the index succeeded")
|
|
||||||
}
|
|
||||||
|
|
||||||
db, err = openScanDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("next scan: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = db.Close() }()
|
|
||||||
|
|
||||||
v, err := userVersion(t.Context(), db)
|
|
||||||
if err != nil || v != schemaVersion {
|
|
||||||
t.Fatalf("userVersion = %d, %v; want %d, nil", v, err, schemaVersion)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -158,11 +87,9 @@ func TestOpenReportDatabaseMissing(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestOpenDatabaseVersionMismatch(t *testing.T) {
|
func TestOpenReportDatabaseVersionMismatch(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// A database stamped with a schema version other than 0 and
|
|
||||||
// schemaVersion. report, trees and scan must all refuse it.
|
|
||||||
path := testDBPath(t)
|
path := testDBPath(t)
|
||||||
|
|
||||||
db, err := openScanDatabase(t.Context(), path)
|
db, err := openScanDatabase(t.Context(), path)
|
||||||
@@ -179,12 +106,7 @@ func TestOpenDatabaseVersionMismatch(t *testing.T) {
|
|||||||
|
|
||||||
_, err = openReportDatabase(t.Context(), path)
|
_, err = openReportDatabase(t.Context(), path)
|
||||||
if !errors.Is(err, errSchemaVersion) {
|
if !errors.Is(err, errSchemaVersion) {
|
||||||
t.Fatalf("report: err = %v, want errSchemaVersion", err)
|
t.Fatalf("err = %v, want errSchemaVersion", err)
|
||||||
}
|
|
||||||
|
|
||||||
_, err = openScanDatabase(t.Context(), path)
|
|
||||||
if !errors.Is(err, errSchemaVersion) {
|
|
||||||
t.Fatalf("scan: err = %v, want errSchemaVersion", err)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -208,49 +130,6 @@ func TestOpenReportDatabaseOK(t *testing.T) {
|
|||||||
_ = db.Close()
|
_ = db.Close()
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestCloseScanDatabaseWhileReportOpen(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
// A report holding the database open stops scan from taking it out
|
|
||||||
// of WAL mode. The -wal and -shm files must then stay beside it, so
|
|
||||||
// that a later report still needs only read access.
|
|
||||||
path := testDBPath(t)
|
|
||||||
|
|
||||||
scanDB, err := openScanDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
reportDB, err := openReportDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
closeScanDatabase(t.Context(), scanDB, path)
|
|
||||||
|
|
||||||
_ = reportDB.Close()
|
|
||||||
|
|
||||||
_, err = os.Stat(path + "-wal")
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("no -wal left: the switch out of WAL mode was not "+
|
|
||||||
"stopped: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
makeReadOnly(t, path)
|
|
||||||
|
|
||||||
reportDB, err = openReportDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("openReportDatabase: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = reportDB.Close() }()
|
|
||||||
|
|
||||||
err = loadFileRows(t.Context(), reportDB, func(scanRec) {})
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("loadFileRows: %v", err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestApplyChangesRoundTrip(t *testing.T) {
|
func TestApplyChangesRoundTrip(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
@@ -261,13 +140,10 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
|||||||
// written.
|
// written.
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{
|
{
|
||||||
size: 2, mtime: time.Unix(20, 999_999_999), head: "h2", tail: "t2",
|
size: 2, mtime: 20, head: "h2", tail: "t2", content: "c2",
|
||||||
content: "c2", path: "/a/tab\tnew\nline",
|
path: "/a/tab\tnew\nline",
|
||||||
},
|
|
||||||
{
|
|
||||||
size: 1, mtime: time.Unix(10, 0), head: "h1", tail: "t1",
|
|
||||||
content: "c1", path: "/a/x",
|
|
||||||
},
|
},
|
||||||
|
{size: 1, mtime: 10, head: "h1", tail: "t1", content: "c1", path: "/a/x"},
|
||||||
}
|
}
|
||||||
|
|
||||||
err := applyChanges(t.Context(), db, recs, nil,
|
err := applyChanges(t.Context(), db, recs, nil,
|
||||||
@@ -276,8 +152,15 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
|||||||
t.Fatalf("applyChanges: %v", err)
|
t.Fatalf("applyChanges: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
// The records come back in path order, which is the order of recs.
|
got, err := loadFileRows(t.Context(), db)
|
||||||
got := dbRecords(t, db)
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
slices.SortFunc(got, func(a, b scanRec) int {
|
||||||
|
return strings.Compare(a.path, b.path)
|
||||||
|
})
|
||||||
|
|
||||||
if !slices.Equal(got, recs) {
|
if !slices.Equal(got, recs) {
|
||||||
t.Fatalf("rows = %+v, want %+v", got, recs)
|
t.Fatalf("rows = %+v, want %+v", got, recs)
|
||||||
}
|
}
|
||||||
@@ -285,8 +168,7 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
|||||||
// An upsert for an existing path updates in place; a delete
|
// An upsert for an existing path updates in place; a delete
|
||||||
// removes exactly its path.
|
// removes exactly its path.
|
||||||
upd := scanRec{
|
upd := scanRec{
|
||||||
size: 3, mtime: time.Unix(30, 0), head: "h3", tail: "t3", content: "c3",
|
size: 3, mtime: 30, head: "h3", tail: "t3", content: "c3", path: "/a/x",
|
||||||
path: "/a/x",
|
|
||||||
}
|
}
|
||||||
|
|
||||||
err = applyChanges(t.Context(), db, []scanRec{upd},
|
err = applyChanges(t.Context(), db, []scanRec{upd},
|
||||||
@@ -295,7 +177,11 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
|||||||
t.Fatalf("applyChanges: %v", err)
|
t.Fatalf("applyChanges: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
got = dbRecords(t, db)
|
got, err = loadFileRows(t.Context(), db)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
if len(got) != 1 || got[0] != upd {
|
if len(got) != 1 || got[0] != upd {
|
||||||
t.Fatalf("rows = %+v, want just %+v", got, upd)
|
t.Fatalf("rows = %+v, want just %+v", got, upd)
|
||||||
}
|
}
|
||||||
@@ -313,7 +199,7 @@ func TestApplyChangesBatching(t *testing.T) {
|
|||||||
recs := make([]scanRec, 0, n)
|
recs := make([]scanRec, 0, n)
|
||||||
for i := range n {
|
for i := range n {
|
||||||
recs = append(recs, scanRec{
|
recs = append(recs, scanRec{
|
||||||
size: int64(i), mtime: time.Unix(1, 0), head: "h", tail: "t",
|
size: int64(i), mtime: 1, head: "h", tail: "t",
|
||||||
path: fmt.Sprintf("/batch/%07d", i),
|
path: fmt.Sprintf("/batch/%07d", i),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
@@ -324,8 +210,9 @@ func TestApplyChangesBatching(t *testing.T) {
|
|||||||
t.Fatalf("applyChanges: %v", err)
|
t.Fatalf("applyChanges: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
if got := dbRecords(t, db); len(got) != n {
|
got, err := loadFileRows(t.Context(), db)
|
||||||
t.Fatalf("records = %d, want %d", len(got), n)
|
if err != nil || len(got) != n {
|
||||||
|
t.Fatalf("loadFileRows = %d rows, %v; want %d", len(got), err, n)
|
||||||
}
|
}
|
||||||
|
|
||||||
deletes := make([]string, 0, n)
|
deletes := make([]string, 0, n)
|
||||||
@@ -339,7 +226,8 @@ func TestApplyChangesBatching(t *testing.T) {
|
|||||||
t.Fatalf("applyChanges deletes: %v", err)
|
t.Fatalf("applyChanges deletes: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
if got := dbRecords(t, db); len(got) != 0 {
|
got, err = loadFileRows(t.Context(), db)
|
||||||
t.Fatalf("records = %d, want 0", len(got))
|
if err != nil || len(got) != 0 {
|
||||||
|
t.Fatalf("loadFileRows = %d rows, %v; want 0", len(got), err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -5,8 +5,6 @@ go 1.25.7
|
|||||||
require (
|
require (
|
||||||
github.com/schollz/progressbar/v3 v3.19.1
|
github.com/schollz/progressbar/v3 v3.19.1
|
||||||
github.com/spf13/cobra v1.10.2
|
github.com/spf13/cobra v1.10.2
|
||||||
golang.org/x/sys v0.46.0
|
|
||||||
golang.org/x/term v0.44.0
|
|
||||||
modernc.org/sqlite v1.54.0
|
modernc.org/sqlite v1.54.0
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -20,6 +18,8 @@ require (
|
|||||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
||||||
github.com/rivo/uniseg v0.4.7 // indirect
|
github.com/rivo/uniseg v0.4.7 // indirect
|
||||||
github.com/spf13/pflag v1.0.9 // indirect
|
github.com/spf13/pflag v1.0.9 // indirect
|
||||||
|
golang.org/x/sys v0.46.0 // indirect
|
||||||
|
golang.org/x/term v0.44.0 // indirect
|
||||||
modernc.org/libc v1.74.1 // indirect
|
modernc.org/libc v1.74.1 // indirect
|
||||||
modernc.org/mathutil v1.7.1 // indirect
|
modernc.org/mathutil v1.7.1 // indirect
|
||||||
modernc.org/memory v1.11.0 // indirect
|
modernc.org/memory v1.11.0 // indirect
|
||||||
|
|||||||
@@ -15,7 +15,6 @@
|
|||||||
// sfdupes scan [--workers N] [-x] PATH...
|
// sfdupes scan [--workers N] [-x] PATH...
|
||||||
// sfdupes report > dupes.tsv
|
// sfdupes report > dupes.tsv
|
||||||
// sfdupes trees > dupetrees.tsv
|
// sfdupes trees > dupetrees.tsv
|
||||||
// sfdupes --version
|
|
||||||
//
|
//
|
||||||
// See README.md for the complete specification.
|
// See README.md for the complete specification.
|
||||||
package main
|
package main
|
||||||
@@ -51,10 +50,6 @@ const (
|
|||||||
// cobra prints for it is the whole message.
|
// cobra prints for it is the whole message.
|
||||||
var errNoSubcommand = errors.New("no subcommand")
|
var errNoSubcommand = errors.New("no subcommand")
|
||||||
|
|
||||||
// errWorkersBelowOne is the usage error for a scan --workers value
|
|
||||||
// below 1.
|
|
||||||
var errWorkersBelowOne = errors.New("--workers must be at least 1")
|
|
||||||
|
|
||||||
// Version is the build version, injected at link time via -ldflags
|
// Version is the build version, injected at link time via -ldflags
|
||||||
// (see the Makefile); "dev" for a plain go build.
|
// (see the Makefile); "dev" for a plain go build.
|
||||||
//
|
//
|
||||||
@@ -62,27 +57,22 @@ var errWorkersBelowOne = errors.New("--workers must be at least 1")
|
|||||||
var Version = "dev"
|
var Version = "dev"
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
// Once the reader of a stdout pipe has gone, as in "sfdupes report |
|
os.Exit(run(os.Args[1:], os.Stderr))
|
||||||
// head", the Go runtime ends the process with SIGPIPE on the next
|
|
||||||
// write instead of returning an error (README "Error handling").
|
|
||||||
// Registering for SIGPIPE with os/signal would change that.
|
|
||||||
os.Exit(run(os.Args[1:], os.Stdout, os.Stderr))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// run executes args against the command tree and returns the process
|
// run executes args against the command tree and returns the process
|
||||||
// exit code. It is the program's single exit point: the subcommands
|
// exit code. It is the program's single exit point: the subcommands
|
||||||
// return their errors instead of exiting, so every deferred cleanup —
|
// return their errors instead of exiting, so every deferred cleanup —
|
||||||
// above all closing the database, which checkpoints the SQLite WAL —
|
// above all closing the database, which checkpoints the SQLite WAL —
|
||||||
// runs before the process ends. The report and trees subcommands write
|
// runs before the process ends.
|
||||||
// their data to stdout.
|
func run(args []string, stderr io.Writer) int {
|
||||||
func run(args []string, stdout, stderr io.Writer) int {
|
|
||||||
// A nil slice makes cobra fall back to os.Args, which would let a
|
// A nil slice makes cobra fall back to os.Args, which would let a
|
||||||
// test binary's own flags reach the command tree.
|
// test binary's own flags reach the command tree.
|
||||||
if args == nil {
|
if args == nil {
|
||||||
args = []string{}
|
args = []string{}
|
||||||
}
|
}
|
||||||
|
|
||||||
root := newRootCommand(stdout, stderr)
|
root := newRootCommand(stderr)
|
||||||
root.SetArgs(args)
|
root.SetArgs(args)
|
||||||
|
|
||||||
err := root.Execute()
|
err := root.Execute()
|
||||||
@@ -92,9 +82,6 @@ func run(args []string, stdout, stderr io.Writer) int {
|
|||||||
switch {
|
switch {
|
||||||
case err == nil:
|
case err == nil:
|
||||||
return exitOK
|
return exitOK
|
||||||
case errors.Is(err, errInterrupted):
|
|
||||||
// The interrupted scan has printed its own line.
|
|
||||||
return exitFatal
|
|
||||||
case errors.As(err, &fatal):
|
case errors.As(err, &fatal):
|
||||||
// The command ran and failed: a runtime error, reported
|
// The command ran and failed: a runtime error, reported
|
||||||
// without the usage text that a usage error gets.
|
// without the usage text that a usage error gets.
|
||||||
@@ -102,35 +89,22 @@ func run(args []string, stdout, stderr io.Writer) int {
|
|||||||
|
|
||||||
return exitFatal
|
return exitFatal
|
||||||
default:
|
default:
|
||||||
// A usage error, which cobra has already reported on stderr.
|
// A usage error: cobra has already printed the message and
|
||||||
|
// the usage text.
|
||||||
return exitUsage
|
return exitUsage
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// newRootCommand builds the command tree. Everything on stdout is
|
// newRootCommand builds the command tree. Everything on stdout is
|
||||||
// machine-readable data, the version line included; all human-facing
|
// machine-readable data; all human-facing output (help, usage, errors)
|
||||||
// output (help, usage, errors) goes to stderr.
|
// goes to stderr.
|
||||||
func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||||
var showVersion bool
|
|
||||||
|
|
||||||
printVersion := runE(func(context.Context, []string) error {
|
|
||||||
_, err := fmt.Fprintf(stdout, "sfdupes %s\n", Version)
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("write stdout: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return nil
|
|
||||||
})
|
|
||||||
|
|
||||||
root := &cobra.Command{
|
root := &cobra.Command{
|
||||||
Use: "sfdupes",
|
Use: "sfdupes",
|
||||||
Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
|
Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
|
||||||
|
Version: Version,
|
||||||
Args: cobra.NoArgs,
|
Args: cobra.NoArgs,
|
||||||
RunE: func(cmd *cobra.Command, args []string) error {
|
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||||
if showVersion {
|
|
||||||
return printVersion(cmd, args)
|
|
||||||
}
|
|
||||||
|
|
||||||
// A missing subcommand prints usage and exits 2: cobra
|
// A missing subcommand prints usage and exits 2: cobra
|
||||||
// prints the usage text for the returned error, and run
|
// prints the usage text for the returned error, and run
|
||||||
// maps everything that is not a fatal error to exit 2.
|
// maps everything that is not a fatal error to exit 2.
|
||||||
@@ -143,11 +117,6 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
|||||||
root.SetErr(stderr)
|
root.SetErr(stderr)
|
||||||
root.CompletionOptions.DisableDefaultCmd = true
|
root.CompletionOptions.DisableDefaultCmd = true
|
||||||
|
|
||||||
// Cobra's built-in version flag prints through the help writer,
|
|
||||||
// stderr; this one prints to stdout.
|
|
||||||
root.Flags().BoolVarP(&showVersion, "version", "v", false,
|
|
||||||
"print the version to stdout")
|
|
||||||
|
|
||||||
var (
|
var (
|
||||||
scanWorkers int
|
scanWorkers int
|
||||||
scanOneFS bool
|
scanOneFS bool
|
||||||
@@ -157,13 +126,7 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
|||||||
Use: cmdScan + " [--workers N] [-x] PATH...",
|
Use: cmdScan + " [--workers N] [-x] PATH...",
|
||||||
Short: "Walk trees and synchronize the scan database",
|
Short: "Walk trees and synchronize the scan database",
|
||||||
Args: cobra.MinimumNArgs(1),
|
Args: cobra.MinimumNArgs(1),
|
||||||
PreRunE: func(cmd *cobra.Command, _ []string) error {
|
|
||||||
return checkScanWorkers(cmd, scanWorkers)
|
|
||||||
},
|
|
||||||
RunE: runE(func(ctx context.Context, args []string) error {
|
RunE: runE(func(ctx context.Context, args []string) error {
|
||||||
ctx, stop := interruptContext(ctx)
|
|
||||||
defer stop()
|
|
||||||
|
|
||||||
return runScan(ctx, args, scanWorkers, scanOneFS)
|
return runScan(ctx, args, scanWorkers, scanOneFS)
|
||||||
}),
|
}),
|
||||||
}
|
}
|
||||||
@@ -177,7 +140,7 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
|||||||
Short: "Read the scan database and print the file-level duplicates report",
|
Short: "Read the scan database and print the file-level duplicates report",
|
||||||
Args: cobra.NoArgs,
|
Args: cobra.NoArgs,
|
||||||
RunE: runE(func(ctx context.Context, _ []string) error {
|
RunE: runE(func(ctx context.Context, _ []string) error {
|
||||||
return runReport(ctx, stdout)
|
return runReport(ctx)
|
||||||
}),
|
}),
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -186,7 +149,7 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
|||||||
Short: "Read the scan database and print the duplicate-tree report",
|
Short: "Read the scan database and print the duplicate-tree report",
|
||||||
Args: cobra.NoArgs,
|
Args: cobra.NoArgs,
|
||||||
RunE: runE(func(ctx context.Context, _ []string) error {
|
RunE: runE(func(ctx context.Context, _ []string) error {
|
||||||
return runTrees(ctx, stdout)
|
return runTrees(ctx)
|
||||||
}),
|
}),
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -195,26 +158,13 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
|||||||
return root
|
return root
|
||||||
}
|
}
|
||||||
|
|
||||||
// checkScanWorkers rejects a scan --workers value below 1. That is a
|
// runE adapts a subcommand implementation to cobra's RunE. Cobra
|
||||||
// usage error reported in one line: cobra prints the returned message
|
// prints the error and the command's usage text for every error RunE
|
||||||
// without the usage text, and run exits 2.
|
// returns, but a subcommand that ran and failed has no usage problem
|
||||||
func checkScanWorkers(cmd *cobra.Command, workers int) error {
|
// to report: both are silenced here, and the error is marked fatal so
|
||||||
if workers >= 1 {
|
// that run reports it on stderr and exits 1 rather than 2. The command's
|
||||||
return nil
|
// context is handed to the implementation: cancelling it unwinds the
|
||||||
}
|
// scan's worker pools.
|
||||||
|
|
||||||
cmd.SilenceUsage = true
|
|
||||||
|
|
||||||
return fmt.Errorf("%w, got %d", errWorkersBelowOne, workers)
|
|
||||||
}
|
|
||||||
|
|
||||||
// runE adapts a subcommand implementation, or the version print, to
|
|
||||||
// cobra's RunE. Cobra prints the error and the command's usage text for
|
|
||||||
// every error RunE returns, but a subcommand that ran and failed has no
|
|
||||||
// usage problem to report: both are silenced here, and the error is
|
|
||||||
// marked fatal so that run reports it on stderr and exits 1 rather than
|
|
||||||
// 2. The command's context is handed to the implementation: cancelling
|
|
||||||
// it unwinds the scan's worker pools.
|
|
||||||
func runE(
|
func runE(
|
||||||
fn func(ctx context.Context, args []string) error,
|
fn func(ctx context.Context, args []string) error,
|
||||||
) func(*cobra.Command, []string) error {
|
) func(*cobra.Command, []string) error {
|
||||||
|
|||||||
+68
-632
@@ -8,7 +8,6 @@ import (
|
|||||||
"io/fs"
|
"io/fs"
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"slices"
|
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
@@ -45,81 +44,23 @@ func assertNoSidecars(t *testing.T, path string) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// makeReadOnly takes write permission away from the database at path,
|
// captureStdout redirects os.Stdout to a file for the rest of the test
|
||||||
// from any WAL sidecar beside it, and from their directory, as for a
|
// and returns a function reading back everything written to it. Only
|
||||||
// user reading a database that a root cron scan keeps. Root ignores
|
// machine-readable data belongs on stdout (README design goal 4), so
|
||||||
// file permissions, so it skips the test when run as root.
|
// the tests assert on it directly.
|
||||||
func makeReadOnly(t *testing.T, path string) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
if os.Geteuid() == 0 {
|
|
||||||
t.Skip("root ignores file permissions")
|
|
||||||
}
|
|
||||||
|
|
||||||
err := os.Chmod(path, 0o400)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
for _, suffix := range walSuffixes {
|
|
||||||
err = os.Chmod(path+suffix, 0o400)
|
|
||||||
if err != nil && !errors.Is(err, fs.ErrNotExist) {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
dir := filepath.Dir(path)
|
|
||||||
|
|
||||||
//nolint:gosec // reaching the database needs the search bit
|
|
||||||
err = os.Chmod(dir, 0o500)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Runs before t.TempDir's own cleanup, which must delete the files.
|
|
||||||
t.Cleanup(func() {
|
|
||||||
//nolint:gosec // removing the directory needs its search bit back
|
|
||||||
_ = os.Chmod(dir, 0o700)
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
// captureStderr redirects os.Stderr to a file for the rest of the test
|
|
||||||
// and returns a function reading back everything written to it. scan
|
|
||||||
// writes its warnings and summary, and report and trees their
|
|
||||||
// summaries, straight to os.Stderr, not to the stderr writer run is
|
|
||||||
// given.
|
|
||||||
func captureStderr(t *testing.T) func() string {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
return capture(t, &os.Stderr)
|
|
||||||
}
|
|
||||||
|
|
||||||
// captureStdout does for os.Stdout what captureStderr does for
|
|
||||||
// os.Stderr. scan is never given run's stdout writer, so anything it
|
|
||||||
// printed would go straight to os.Stdout. The scan tests pass os.Stdout
|
|
||||||
// as run's stdout too, so the one capture sees both.
|
|
||||||
func captureStdout(t *testing.T) func() string {
|
func captureStdout(t *testing.T) func() string {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
return capture(t, &os.Stdout)
|
f, err := os.Create(filepath.Join(t.TempDir(), "stdout"))
|
||||||
}
|
|
||||||
|
|
||||||
// capture redirects *std, which is os.Stdout or os.Stderr, to a file
|
|
||||||
// for the rest of the test and returns a function reading back
|
|
||||||
// everything written to it.
|
|
||||||
func capture(t *testing.T, std **os.File) func() string {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
f, err := os.Create(filepath.Join(t.TempDir(), "output"))
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
|
|
||||||
saved := *std
|
saved := os.Stdout
|
||||||
*std = f
|
os.Stdout = f
|
||||||
|
|
||||||
t.Cleanup(func() {
|
t.Cleanup(func() {
|
||||||
*std = saved
|
os.Stdout = saved
|
||||||
|
|
||||||
_ = f.Close()
|
_ = f.Close()
|
||||||
})
|
})
|
||||||
@@ -150,13 +91,13 @@ func capture(t *testing.T, std **os.File) func() string {
|
|||||||
// brokenDatabase writes a database that opens cleanly and passes the
|
// brokenDatabase writes a database that opens cleanly and passes the
|
||||||
// schema-version check but has no files table, so the first query
|
// schema-version check but has no files table, so the first query
|
||||||
// fails with the database already open: a fatal error on a path that
|
// fails with the database already open: a fatal error on a path that
|
||||||
// owns an open database. It closes the database the way scan does.
|
// owns an open database.
|
||||||
func brokenDatabase(t *testing.T) string {
|
func brokenDatabase(t *testing.T) string {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
path := testDBPath(t)
|
path := testDBPath(t)
|
||||||
|
|
||||||
db, err := openDB(path, scanParams)
|
db, err := openDB(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
@@ -167,7 +108,10 @@ func brokenDatabase(t *testing.T) string {
|
|||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
|
|
||||||
closeScanDatabase(t.Context(), db, path)
|
err = db.Close()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
return path
|
return path
|
||||||
}
|
}
|
||||||
@@ -200,31 +144,27 @@ func TestOpenDatabaseKeepsWALWhileOpen(t *testing.T) {
|
|||||||
|
|
||||||
func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
|
func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
|
||||||
// Every subcommand that owns an open database must close it when
|
// Every subcommand that owns an open database must close it when
|
||||||
// it fails: no os.Exit between the open and the return. The
|
// it fails: no os.Exit between the open and the return.
|
||||||
// sidecar check is evidence of the close only for scan: report and
|
cases := map[string][]string{
|
||||||
// trees only read a database that is out of WAL mode, which leaves
|
cmdScan: {cmdScan},
|
||||||
// nothing on disk whether they close it or not.
|
cmdReport: {cmdReport},
|
||||||
//
|
cmdTrees: {cmdTrees},
|
||||||
// The subcommands come from the command tree, so a new one is
|
}
|
||||||
// checked too: one wired with a bare RunE instead of runE reports
|
|
||||||
// its failure as a usage error, exit 2 with the usage text.
|
|
||||||
for _, cmd := range newRootCommand(io.Discard, io.Discard).Commands() {
|
|
||||||
name := cmd.Name()
|
|
||||||
|
|
||||||
|
for name, args := range cases {
|
||||||
t.Run(name, func(t *testing.T) {
|
t.Run(name, func(t *testing.T) {
|
||||||
path := brokenDatabase(t)
|
path := brokenDatabase(t)
|
||||||
t.Setenv(databaseEnv, path)
|
t.Setenv(databaseEnv, path)
|
||||||
|
|
||||||
args := []string{name}
|
|
||||||
if name == cmdScan {
|
if name == cmdScan {
|
||||||
args = append(args, t.TempDir())
|
args = append(args, t.TempDir())
|
||||||
}
|
}
|
||||||
|
|
||||||
stdout := captureStdout(t)
|
|
||||||
|
|
||||||
var stderr bytes.Buffer
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
code := run(args, os.Stdout, &stderr)
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
|
code := run(args, &stderr)
|
||||||
if code != exitFatal {
|
if code != exitFatal {
|
||||||
t.Errorf("run(%v) = %d, want %d", args, code, exitFatal)
|
t.Errorf("run(%v) = %d, want %d", args, code, exitFatal)
|
||||||
}
|
}
|
||||||
@@ -242,19 +182,19 @@ func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRunNonexistentPathIsFatalNotUsage(t *testing.T) {
|
func TestRunMissingOperandIsFatalNotUsage(t *testing.T) {
|
||||||
// README §Error handling: a PATH operand that does not exist is a
|
// README §Error handling: a PATH operand that does not exist is a
|
||||||
// fatal error (1), not a usage error (2) — and a runtime failure
|
// fatal error (1), not a usage error (2) — and a runtime failure
|
||||||
// must not dump the usage text.
|
// must not dump the usage text.
|
||||||
t.Setenv(databaseEnv, testDBPath(t))
|
t.Setenv(databaseEnv, testDBPath(t))
|
||||||
|
|
||||||
stdout := captureStdout(t)
|
|
||||||
|
|
||||||
var stderr bytes.Buffer
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
missing := filepath.Join(t.TempDir(), "nope")
|
missing := filepath.Join(t.TempDir(), "nope")
|
||||||
|
|
||||||
code := run([]string{cmdScan, missing}, os.Stdout, &stderr)
|
code := run([]string{cmdScan, missing}, &stderr)
|
||||||
if code != exitFatal {
|
if code != exitFatal {
|
||||||
t.Errorf("run(scan %s) = %d, want %d", missing, code, exitFatal)
|
t.Errorf("run(scan %s) = %d, want %d", missing, code, exitFatal)
|
||||||
}
|
}
|
||||||
@@ -283,34 +223,6 @@ func assertFatalOutput(t *testing.T, stderr, stdout string) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRunMissingDatabaseIsFatal(t *testing.T) {
|
|
||||||
// README §Database: report and trees need an existing database; a
|
|
||||||
// missing one exits 1 with a message telling the user to run scan.
|
|
||||||
for _, name := range []string{cmdReport, cmdTrees} {
|
|
||||||
t.Run(name, func(t *testing.T) {
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{name}, &stdout, &stderr)
|
|
||||||
if code != exitFatal {
|
|
||||||
t.Errorf("run(%s) = %d, want %d", name, code, exitFatal)
|
|
||||||
}
|
|
||||||
|
|
||||||
want := "sfdupes: " + path + ": no database (run \"sfdupes " +
|
|
||||||
"scan\" first, or set " + databaseEnv + ")\n"
|
|
||||||
if got := stderr.String(); got != want {
|
|
||||||
t.Errorf("stderr = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
if got := stdout.String(); got != "" {
|
|
||||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunUsageErrors(t *testing.T) {
|
func TestRunUsageErrors(t *testing.T) {
|
||||||
// Usage errors keep exiting 2 with cobra's own report on stderr.
|
// Usage errors keep exiting 2 with cobra's own report on stderr.
|
||||||
cases := map[string]struct {
|
cases := map[string]struct {
|
||||||
@@ -331,9 +243,11 @@ func TestRunUsageErrors(t *testing.T) {
|
|||||||
// path that does not exist.
|
// path that does not exist.
|
||||||
t.Setenv(databaseEnv, testDBPath(t))
|
t.Setenv(databaseEnv, testDBPath(t))
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
code := run(tc.args, &stdout, &stderr)
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
|
code := run(tc.args, &stderr)
|
||||||
if code != exitUsage {
|
if code != exitUsage {
|
||||||
t.Errorf("run(%v) = %d, want %d", tc.args, code, exitUsage)
|
t.Errorf("run(%v) = %d, want %d", tc.args, code, exitUsage)
|
||||||
}
|
}
|
||||||
@@ -342,115 +256,43 @@ func TestRunUsageErrors(t *testing.T) {
|
|||||||
t.Errorf("stderr = %q, want %q", stderr.String(), tc.want)
|
t.Errorf("stderr = %q, want %q", stderr.String(), tc.want)
|
||||||
}
|
}
|
||||||
|
|
||||||
if got := stdout.String(); got != "" {
|
if got := stdout(); got != "" {
|
||||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
t.Errorf("stdout = %q, want nothing (data only)", got)
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRunScanRejectsWorkersBelowOne(t *testing.T) {
|
// TestRunHelpAndVersionSucceed checks that the two informational flags
|
||||||
// README §scan mode: --workers below 1 is a usage error reported in
|
// exit 0 and keep their human-facing output on stderr.
|
||||||
// one line on stderr, before the scan opens the database.
|
//
|
||||||
for _, workers := range []string{"0", "-1"} {
|
//nolint:paralleltest // captureStdout replaces the process-wide os.Stdout
|
||||||
t.Run(workers, func(t *testing.T) {
|
func TestRunHelpAndVersionSucceed(t *testing.T) {
|
||||||
dbPath := testDBPath(t)
|
assertHumanOutput(t, "--help")
|
||||||
t.Setenv(databaseEnv, dbPath)
|
assertHumanOutput(t, "--version")
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
args := []string{cmdScan, "--workers", workers, t.TempDir()}
|
|
||||||
|
|
||||||
code := run(args, &stdout, &stderr)
|
|
||||||
if code != exitUsage {
|
|
||||||
t.Errorf("run(%v) = %d, want %d", args, code, exitUsage)
|
|
||||||
}
|
|
||||||
|
|
||||||
want := "Error: --workers must be at least 1, got " + workers +
|
|
||||||
"\n"
|
|
||||||
if got := stderr.String(); got != want {
|
|
||||||
t.Errorf("stderr = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
if got := stdout.String(); got != "" {
|
|
||||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err := os.Stat(dbPath)
|
|
||||||
if !errors.Is(err, fs.ErrNotExist) {
|
|
||||||
t.Errorf("stat %s: %v, want the database never created",
|
|
||||||
dbPath, err)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRunHelp(t *testing.T) {
|
// assertHumanOutput runs sfdupes with one informational flag and checks
|
||||||
t.Parallel()
|
// that it succeeds with its output on stderr and stdout untouched
|
||||||
|
// (README design goal 4).
|
||||||
|
func assertHumanOutput(t *testing.T, arg string) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
// README §Subcommands: help goes to stderr, exits 0, and leaves
|
var stderr bytes.Buffer
|
||||||
// stdout empty.
|
|
||||||
cases := [][]string{{"--help"}, {"-h"}, {cmdScan, "--help"}}
|
|
||||||
|
|
||||||
for _, args := range cases {
|
stdout := captureStdout(t)
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run(args, &stdout, &stderr)
|
code := run([]string{arg}, &stderr)
|
||||||
if code != exitOK {
|
|
||||||
t.Errorf("run(%v) = %d, want %d", args, code, exitOK)
|
|
||||||
}
|
|
||||||
|
|
||||||
if !strings.Contains(stderr.String(), usageMarker) {
|
|
||||||
t.Errorf("run(%v) stderr = %q, want the help text",
|
|
||||||
args, stderr.String())
|
|
||||||
}
|
|
||||||
|
|
||||||
if got := stdout.String(); got != "" {
|
|
||||||
t.Errorf("run(%v) stdout = %q, want nothing (data only)",
|
|
||||||
args, got)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunVersion(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
// README §Subcommands: the version is one line on stdout, with
|
|
||||||
// nothing on stderr, and exits 0.
|
|
||||||
for _, arg := range []string{"--version", "-v"} {
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{arg}, &stdout, &stderr)
|
|
||||||
if code != exitOK {
|
if code != exitOK {
|
||||||
t.Errorf("run(%s) = %d, want %d", arg, code, exitOK)
|
t.Errorf("run(%s) = %d, want %d", arg, code, exitOK)
|
||||||
}
|
}
|
||||||
|
|
||||||
want := "sfdupes " + Version + "\n"
|
if stderr.Len() == 0 {
|
||||||
if got := stdout.String(); got != want {
|
t.Errorf("run(%s) wrote nothing to stderr", arg)
|
||||||
t.Errorf("run(%s) stdout = %q, want %q", arg, got, want)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
if got := stderr.String(); got != "" {
|
if got := stdout(); got != "" {
|
||||||
t.Errorf("run(%s) stderr = %q, want nothing", arg, got)
|
t.Errorf("stdout = %q, want nothing (data only)", got)
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunVersionWriteFailureIsFatal(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
// README §Error handling: a stdout write failure exits 1, reported
|
|
||||||
// in one line on stderr.
|
|
||||||
var stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{"--version"}, failingWriter{}, &stderr)
|
|
||||||
if code != exitFatal {
|
|
||||||
t.Errorf("run(--version) = %d, want %d", code, exitFatal)
|
|
||||||
}
|
|
||||||
|
|
||||||
want := "sfdupes: write stdout: " + errWriteFailed.Error() + "\n"
|
|
||||||
if got := stderr.String(); got != want {
|
|
||||||
t.Errorf("stderr = %q, want %q", got, want)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -478,195 +320,52 @@ func scanFixture(t *testing.T) []string {
|
|||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
|
|
||||||
scanOK(t, dir)
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
return dupes
|
|
||||||
}
|
|
||||||
|
|
||||||
// scanOK runs scan over operands, fails the test unless it exits 0 with
|
|
||||||
// nothing on stdout, and returns everything it printed to stderr.
|
|
||||||
func scanOK(t *testing.T, operands ...string) string {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
stdout := captureStdout(t)
|
stdout := captureStdout(t)
|
||||||
stderr := captureStderr(t)
|
|
||||||
|
|
||||||
code := run(append([]string{cmdScan}, operands...), os.Stdout, os.Stderr)
|
code := run([]string{cmdScan, dir}, &stderr)
|
||||||
if code != exitOK {
|
if code != exitOK {
|
||||||
t.Fatalf("run(scan %q) = %d, want %d; stderr: %s",
|
t.Fatalf("run(scan) = %d, want %d; stderr: %s",
|
||||||
operands, code, exitOK, stderr())
|
code, exitOK, stderr.String())
|
||||||
}
|
}
|
||||||
|
|
||||||
if got := stdout(); got != "" {
|
if got := stdout(); got != "" {
|
||||||
t.Errorf("scan stdout = %q, want nothing (data only)", got)
|
t.Errorf("scan stdout = %q, want nothing (data only)", got)
|
||||||
}
|
}
|
||||||
|
|
||||||
return stderr()
|
return dupes
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
||||||
// README §Error handling: a scan that skips a file it cannot read
|
|
||||||
// warns, counts the skip in its summary, and still exits 0, which
|
|
||||||
// scanOK checks along with the empty stdout.
|
|
||||||
if os.Geteuid() == 0 {
|
|
||||||
t.Skip("root ignores file permissions")
|
|
||||||
}
|
|
||||||
|
|
||||||
path := testDBPath(t)
|
path := testDBPath(t)
|
||||||
t.Setenv(databaseEnv, path)
|
t.Setenv(databaseEnv, path)
|
||||||
|
|
||||||
dir := t.TempDir()
|
scanFixture(t)
|
||||||
writeFile(t, dir, "a.bin", pattern(1, 300))
|
|
||||||
|
|
||||||
// Same size as a.bin, so the scan reads it, and the read fails.
|
|
||||||
unreadable := writeFile(t, dir, "unreadable.bin", pattern(2, 300))
|
|
||||||
|
|
||||||
err := os.Chmod(unreadable, 0)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
stderr := scanOK(t, dir)
|
|
||||||
|
|
||||||
warning := "hash " + unreadable + ": open " + unreadable +
|
|
||||||
": permission denied\n"
|
|
||||||
if !strings.Contains(stderr, warning) {
|
|
||||||
t.Errorf("stderr = %q, want %q", stderr, warning)
|
|
||||||
}
|
|
||||||
|
|
||||||
summary := "scan: 1 files seen (1 added, 0 updated, 0 removed, " +
|
|
||||||
"0 unchanged), 1 skipped\n"
|
|
||||||
if !strings.Contains(stderr, summary) {
|
|
||||||
t.Errorf("stderr = %q, want %q", stderr, summary)
|
|
||||||
}
|
|
||||||
|
|
||||||
assertNoSidecars(t, path)
|
assertNoSidecars(t, path)
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRunScanSkipsSymlinkOperand(t *testing.T) {
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
dir := t.TempDir()
|
|
||||||
writeFile(t, dir, "target/sub/f", pattern(1, 10))
|
|
||||||
|
|
||||||
link := filepath.Join(dir, "link")
|
|
||||||
|
|
||||||
err := os.Symlink(filepath.Join(dir, "target"), link)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Scanning a directory through the symlink stores a record beneath
|
|
||||||
// the symlink's own path for a file beneath its target.
|
|
||||||
scanOK(t, filepath.Join(link, "sub"))
|
|
||||||
|
|
||||||
assertOperandSkipped(t, path, link, "symlink",
|
|
||||||
filepath.Join(link, "sub", "f"))
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunScanWalksOperandUnderSymlinkOperand(t *testing.T) {
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
dir := t.TempDir()
|
|
||||||
writeFile(t, dir, "target/sub/f", pattern(1, 10))
|
|
||||||
|
|
||||||
link := filepath.Join(dir, "link")
|
|
||||||
|
|
||||||
err := os.Symlink(filepath.Join(dir, "target"), link)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// link is dropped as a symlink, but link/sub must still be scanned,
|
|
||||||
// not dropped as lying under link.
|
|
||||||
scanOK(t, link, filepath.Join(link, "sub"))
|
|
||||||
|
|
||||||
db, err := openDB(path, reportParams)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
t.Cleanup(func() { _ = db.Close() })
|
|
||||||
|
|
||||||
recordByPath(t, dbRecords(t, db), filepath.Join(link, "sub", "f"))
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunScanSkipsZFSOperand(t *testing.T) {
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
zfs := filepath.Join(t.TempDir(), ".zfs")
|
|
||||||
snapshot := filepath.Join(zfs, "snapshot", "hourly")
|
|
||||||
f := writeFile(t, snapshot, "f", pattern(1, 10))
|
|
||||||
|
|
||||||
// An operand beneath a .zfs directory is walked, because it is not
|
|
||||||
// itself named .zfs.
|
|
||||||
scanOK(t, snapshot)
|
|
||||||
|
|
||||||
assertOperandSkipped(t, path, zfs, ".zfs directory", f)
|
|
||||||
}
|
|
||||||
|
|
||||||
// assertOperandSkipped scans operand alone and checks that it is skipped
|
|
||||||
// as kind: a warning naming it, one skip in the summary, exit 0, and the
|
|
||||||
// record for kept, which an earlier scan stored beneath operand, still
|
|
||||||
// in the database at dbPath.
|
|
||||||
func assertOperandSkipped(t *testing.T, dbPath, operand, kind,
|
|
||||||
kept string,
|
|
||||||
) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
stderr := scanOK(t, operand)
|
|
||||||
|
|
||||||
warning := "walk " + operand + ": skipping " + kind + " operand\n"
|
|
||||||
if !strings.Contains(stderr, warning) {
|
|
||||||
t.Errorf("stderr = %q, want %q", stderr, warning)
|
|
||||||
}
|
|
||||||
|
|
||||||
summary := "scan: 0 files seen (0 added, 0 updated, 0 removed, " +
|
|
||||||
"0 unchanged), 1 skipped\n"
|
|
||||||
if !strings.Contains(stderr, summary) {
|
|
||||||
t.Errorf("stderr = %q, want %q", stderr, summary)
|
|
||||||
}
|
|
||||||
|
|
||||||
db, err := openDB(dbPath, reportParams)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
t.Cleanup(func() { _ = db.Close() })
|
|
||||||
|
|
||||||
recordByPath(t, dbRecords(t, db), kept)
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunReportSucceeds(t *testing.T) {
|
func TestRunReportSucceeds(t *testing.T) {
|
||||||
path := testDBPath(t)
|
path := testDBPath(t)
|
||||||
t.Setenv(databaseEnv, path)
|
t.Setenv(databaseEnv, path)
|
||||||
|
|
||||||
dupes := scanFixture(t)
|
dupes := scanFixture(t)
|
||||||
|
|
||||||
var stdout bytes.Buffer
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
stderr := captureStderr(t)
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
code := run([]string{cmdReport}, &stdout, os.Stderr)
|
code := run([]string{cmdReport}, &stderr)
|
||||||
if code != exitOK {
|
if code != exitOK {
|
||||||
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
||||||
code, exitOK, stderr())
|
code, exitOK, stderr.String())
|
||||||
}
|
}
|
||||||
|
|
||||||
want := "first\tdupe\tsize\n" + dupes[0] + "\t" + dupes[1] + "\t300\n"
|
want := "first\tdupe\tsize\n" + dupes[0] + "\t" + dupes[1] + "\t300\n"
|
||||||
if got := stdout.String(); got != want {
|
if got := stdout(); got != want {
|
||||||
t.Errorf("stdout = %q, want %q", got, want)
|
t.Errorf("stdout = %q, want %q", got, want)
|
||||||
}
|
}
|
||||||
|
|
||||||
want = "report: 2 records read, 1 duplicate groups, 1 dupe files, " +
|
|
||||||
"300 B reclaimable\n"
|
|
||||||
if got := stderr(); got != want {
|
|
||||||
t.Errorf("stderr = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
assertNoSidecars(t, path)
|
assertNoSidecars(t, path)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -676,286 +375,23 @@ func TestRunTreesSucceeds(t *testing.T) {
|
|||||||
|
|
||||||
dupes := scanFixture(t)
|
dupes := scanFixture(t)
|
||||||
|
|
||||||
var stdout bytes.Buffer
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
stderr := captureStderr(t)
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
code := run([]string{cmdTrees}, &stdout, os.Stderr)
|
code := run([]string{cmdTrees}, &stderr)
|
||||||
if code != exitOK {
|
if code != exitOK {
|
||||||
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
||||||
code, exitOK, stderr())
|
code, exitOK, stderr.String())
|
||||||
}
|
}
|
||||||
|
|
||||||
// The two directories holding the duplicate pair are duplicate
|
// The two directories holding the duplicate pair are duplicate
|
||||||
// trees of each other.
|
// trees of each other.
|
||||||
want := "first\tdupe\tfiles\tsize\n" +
|
want := "first\tdupe\tfiles\tsize\n" +
|
||||||
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) + "\t1\t300\n"
|
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) + "\t1\t300\n"
|
||||||
if got := stdout.String(); got != want {
|
if got := stdout(); got != want {
|
||||||
t.Errorf("stdout = %q, want %q", got, want)
|
t.Errorf("stdout = %q, want %q", got, want)
|
||||||
}
|
}
|
||||||
|
|
||||||
want = "trees: 2 records read, 1 duplicate tree groups, 1 dupe trees, " +
|
|
||||||
"300 B reclaimable\n"
|
|
||||||
if got := stderr(); got != want {
|
|
||||||
t.Errorf("stderr = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
assertNoSidecars(t, path)
|
assertNoSidecars(t, path)
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRunReportsNeedOnlyReadAccess(t *testing.T) {
|
|
||||||
// README §Database: report and trees need only read access to the
|
|
||||||
// database file. With its directory read-only as well, SQLite
|
|
||||||
// cannot create any file beside it.
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
dupes := scanFixture(t)
|
|
||||||
assertNoSidecars(t, path)
|
|
||||||
makeReadOnly(t, path)
|
|
||||||
|
|
||||||
cases := map[string]string{
|
|
||||||
cmdReport: "first\tdupe\tsize\n" +
|
|
||||||
dupes[0] + "\t" + dupes[1] + "\t300\n",
|
|
||||||
cmdTrees: "first\tdupe\tfiles\tsize\n" +
|
|
||||||
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) +
|
|
||||||
"\t1\t300\n",
|
|
||||||
}
|
|
||||||
|
|
||||||
for name, want := range cases {
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{name}, &stdout, &stderr)
|
|
||||||
if code != exitOK {
|
|
||||||
t.Errorf("run(%s) = %d, want %d; stderr: %s",
|
|
||||||
name, code, exitOK, stderr.String())
|
|
||||||
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
if got := stdout.String(); got != want {
|
|
||||||
t.Errorf("%s stdout = %q, want %q", name, got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// assertRunsUseDatabase runs scan, then report and trees, against the
|
|
||||||
// database that SFDUPES_DATABASE names, the file name in dir. It fails
|
|
||||||
// unless the reports find the duplicate pair the scan recorded and dir
|
|
||||||
// then holds only that file and its lock file: nothing was created
|
|
||||||
// under a shortened name.
|
|
||||||
func assertRunsUseDatabase(t *testing.T, dir, name string) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
dupes := scanFixture(t)
|
|
||||||
|
|
||||||
want := "first\tdupe\tsize\n" + dupes[0] + "\t" + dupes[1] + "\t300\n"
|
|
||||||
if got := runStdout(t, cmdReport); got != want {
|
|
||||||
t.Errorf("report stdout = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
want = "first\tdupe\tfiles\tsize\n" +
|
|
||||||
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) + "\t1\t300\n"
|
|
||||||
if got := runStdout(t, cmdTrees); got != want {
|
|
||||||
t.Errorf("trees stdout = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
entries, err := os.ReadDir(dir)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
got := make([]string, 0, len(entries))
|
|
||||||
for _, e := range entries {
|
|
||||||
got = append(got, e.Name())
|
|
||||||
}
|
|
||||||
|
|
||||||
if wantFiles := []string{name, name + ".lock"}; !slices.Equal(got, wantFiles) {
|
|
||||||
t.Errorf("%s holds %q, want %q", dir, got, wantFiles)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunDatabasePathUsedAsGiven(t *testing.T) {
|
|
||||||
// README §Database: the path names the database file exactly. In
|
|
||||||
// SQLite's connection string an unescaped ? or # would end the file
|
|
||||||
// name and % would start an escape, and a path starting with //
|
|
||||||
// could be read as a host name. %25 is a valid escape, so unescaped
|
|
||||||
// this name opens a file named a without any error.
|
|
||||||
const name = "a?b#c%25d e.sqlite"
|
|
||||||
|
|
||||||
t.Run("absolute", func(t *testing.T) {
|
|
||||||
dir := t.TempDir()
|
|
||||||
t.Setenv(databaseEnv, filepath.Join(dir, name))
|
|
||||||
|
|
||||||
assertRunsUseDatabase(t, dir, name)
|
|
||||||
})
|
|
||||||
|
|
||||||
t.Run("leading double slash", func(t *testing.T) {
|
|
||||||
dir := t.TempDir()
|
|
||||||
t.Setenv(databaseEnv, "/"+filepath.Join(dir, name))
|
|
||||||
|
|
||||||
assertRunsUseDatabase(t, dir, name)
|
|
||||||
})
|
|
||||||
|
|
||||||
t.Run("relative", func(t *testing.T) {
|
|
||||||
dir := t.TempDir()
|
|
||||||
t.Chdir(dir)
|
|
||||||
t.Setenv(databaseEnv, name)
|
|
||||||
|
|
||||||
assertRunsUseDatabase(t, dir, name)
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
// holdScanLock takes the lock on the database at path, as a running
|
|
||||||
// scan does, and holds it until the test ends. It fails the test when
|
|
||||||
// the lock is already held.
|
|
||||||
func holdScanLock(t *testing.T, path string) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
lock, err := lockScanDatabase(path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("lock %s: %v", path, err)
|
|
||||||
}
|
|
||||||
|
|
||||||
t.Cleanup(func() { _ = lock.Close() })
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunSecondScanFails(t *testing.T) {
|
|
||||||
// README §Database: while one scan holds the lock, a second scan
|
|
||||||
// fails at once, naming the lock file, without creating the
|
|
||||||
// database.
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
holdScanLock(t, path)
|
|
||||||
|
|
||||||
stdout := captureStdout(t)
|
|
||||||
|
|
||||||
var stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{cmdScan, t.TempDir()}, os.Stdout, &stderr)
|
|
||||||
if code != exitFatal {
|
|
||||||
t.Errorf("run(scan) = %d, want %d", code, exitFatal)
|
|
||||||
}
|
|
||||||
|
|
||||||
want := "sfdupes: another scan is running (lock held on " +
|
|
||||||
path + ".lock)\n"
|
|
||||||
if got := stderr.String(); got != want {
|
|
||||||
t.Errorf("stderr = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
if got := stdout(); got != "" {
|
|
||||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err := os.Stat(path)
|
|
||||||
if !errors.Is(err, fs.ErrNotExist) {
|
|
||||||
t.Errorf("stat %s = %v, want the database not created", path, err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunScanReleasesLock(t *testing.T) {
|
|
||||||
// README §Database: a scan releases the lock however it ends.
|
|
||||||
t.Run("success", func(t *testing.T) {
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
scanFixture(t)
|
|
||||||
holdScanLock(t, path)
|
|
||||||
})
|
|
||||||
|
|
||||||
t.Run("fatal error", func(t *testing.T) {
|
|
||||||
path := brokenDatabase(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
code := run([]string{cmdScan, t.TempDir()}, io.Discard, io.Discard)
|
|
||||||
if code != exitFatal {
|
|
||||||
t.Fatalf("run(scan) = %d, want %d", code, exitFatal)
|
|
||||||
}
|
|
||||||
|
|
||||||
holdScanLock(t, path)
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunReportsDuringScan(t *testing.T) {
|
|
||||||
// README §Database: report and trees never take the lock, so they
|
|
||||||
// run while a scan holds it.
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
scanFixture(t)
|
|
||||||
holdScanLock(t, path)
|
|
||||||
|
|
||||||
for _, name := range []string{cmdReport, cmdTrees} {
|
|
||||||
var stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{name}, io.Discard, &stderr)
|
|
||||||
if code != exitOK {
|
|
||||||
t.Errorf("run(%s) = %d, want %d; stderr: %s",
|
|
||||||
name, code, exitOK, stderr.String())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunStdoutClosedIsFatal(t *testing.T) {
|
|
||||||
// README §Error handling: a stdout write failure exits 1, reported
|
|
||||||
// in one line on stderr.
|
|
||||||
for _, name := range []string{cmdReport, cmdTrees} {
|
|
||||||
t.Run(name, func(t *testing.T) {
|
|
||||||
t.Setenv(databaseEnv, testDBPath(t))
|
|
||||||
|
|
||||||
scanFixture(t)
|
|
||||||
|
|
||||||
stdout, err := os.Create(filepath.Join(t.TempDir(), "stdout"))
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = stdout.Close()
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
var stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{name}, stdout, &stderr)
|
|
||||||
if code != exitFatal {
|
|
||||||
t.Errorf("run(%s) = %d, want %d", name, code, exitFatal)
|
|
||||||
}
|
|
||||||
|
|
||||||
got := stderr.String()
|
|
||||||
if !strings.HasPrefix(got, "sfdupes: write stdout: ") ||
|
|
||||||
!strings.Contains(got, os.ErrClosed.Error()) ||
|
|
||||||
strings.Count(got, "\n") != 1 {
|
|
||||||
t.Errorf("stderr = %q, want one line reporting the "+
|
|
||||||
"failed stdout write", got)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// errWriteFailed is the error failingWriter returns.
|
|
||||||
var errWriteFailed = errors.New("write failed")
|
|
||||||
|
|
||||||
// failingWriter is a stdout that fails every write.
|
|
||||||
type failingWriter struct{}
|
|
||||||
|
|
||||||
func (failingWriter) Write([]byte) (int, error) { return 0, errWriteFailed }
|
|
||||||
|
|
||||||
func TestStdoutWriteErrorPropagates(t *testing.T) {
|
|
||||||
t.Setenv(databaseEnv, testDBPath(t))
|
|
||||||
|
|
||||||
scanFixture(t)
|
|
||||||
|
|
||||||
cases := map[string]func(context.Context, io.Writer) error{
|
|
||||||
cmdReport: runReport,
|
|
||||||
cmdTrees: runTrees,
|
|
||||||
}
|
|
||||||
|
|
||||||
for name, fn := range cases {
|
|
||||||
err := fn(t.Context(), failingWriter{})
|
|
||||||
if !errors.Is(err, errWriteFailed) {
|
|
||||||
t.Errorf("%s: error = %v, want %v", name, err, errWriteFailed)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -1,5 +0,0 @@
|
|||||||
{
|
|
||||||
"devDependencies": {
|
|
||||||
"prettier": "3.8.1"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
+17
-46
@@ -6,7 +6,6 @@ import (
|
|||||||
"time"
|
"time"
|
||||||
|
|
||||||
"github.com/schollz/progressbar/v3"
|
"github.com/schollz/progressbar/v3"
|
||||||
"golang.org/x/term"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// plainInterval is the minimum time between progress lines when stderr
|
// plainInterval is the minimum time between progress lines when stderr
|
||||||
@@ -25,23 +24,24 @@ const percentScale = 100
|
|||||||
|
|
||||||
// stderrIsTTY reports whether stderr is attached to a terminal.
|
// stderrIsTTY reports whether stderr is attached to a terminal.
|
||||||
func stderrIsTTY() bool {
|
func stderrIsTTY() bool {
|
||||||
return term.IsTerminal(int(os.Stderr.Fd()))
|
fi, err := os.Stderr.Stat()
|
||||||
|
if err != nil {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
return fi.Mode()&os.ModeCharDevice != 0
|
||||||
}
|
}
|
||||||
|
|
||||||
// progress renders one scan pass's progress on stderr. On a TTY it
|
// progress renders one scan pass's progress on stderr. On a TTY it
|
||||||
// delegates to the progressbar library (spinner style when the total is
|
// delegates to the progressbar library (spinner style when the total is
|
||||||
// unknown, full bar with count/percent/rate/elapsed/ETA otherwise). When
|
// unknown, full bar with count/percent/rate/elapsed/ETA otherwise). When
|
||||||
// stderr is not a TTY it emits no ANSI redraws: it prints a plain
|
// stderr is not a TTY it emits no ANSI redraws: it prints a plain
|
||||||
// one-line update as the pass starts, then no more often than every
|
// one-line update no more often than every plainInterval.
|
||||||
// plainInterval.
|
|
||||||
//
|
//
|
||||||
// All methods must be called from the main goroutine only. On a TTY
|
// All methods must be called from the main goroutine only. A nil
|
||||||
// the library also redraws a spinner from its own goroutine, several
|
// *progress is a valid no-display receiver: every method is a no-op,
|
||||||
// times a second, so its count and elapsed time stay current while a
|
// so batched database flushes during the streaming pass can reuse the
|
||||||
// pass waits for its next item. A nil *progress is a valid
|
// update-pass helpers without rendering anything.
|
||||||
// no-display receiver: every method is a no-op, so batched database
|
|
||||||
// flushes during the streaming pass can reuse the update-pass helpers
|
|
||||||
// without rendering anything.
|
|
||||||
type progress struct {
|
type progress struct {
|
||||||
label string
|
label string
|
||||||
total int64 // -1 when unknown (walk pass)
|
total int64 // -1 when unknown (walk pass)
|
||||||
@@ -53,22 +53,10 @@ type progress struct {
|
|||||||
|
|
||||||
func newProgress(label string, total int64) *progress {
|
func newProgress(label string, total int64) *progress {
|
||||||
p := &progress{label: label, total: total, start: time.Now()}
|
p := &progress{label: label, total: total, start: time.Now()}
|
||||||
if stderrIsTTY() {
|
if !stderrIsTTY() {
|
||||||
p.bar = newBar(label, total)
|
|
||||||
|
|
||||||
return p
|
return p
|
||||||
}
|
}
|
||||||
|
|
||||||
// Print the zero state at once: the first item may take minutes,
|
|
||||||
// and a pass must never look hung.
|
|
||||||
p.last = p.start
|
|
||||||
fmt.Fprintln(os.Stderr, p.plainLine())
|
|
||||||
|
|
||||||
return p
|
|
||||||
}
|
|
||||||
|
|
||||||
// newBar builds the TTY display for newProgress.
|
|
||||||
func newBar(label string, total int64) *progressbar.ProgressBar {
|
|
||||||
opts := []progressbar.Option{
|
opts := []progressbar.Option{
|
||||||
progressbar.OptionSetWriter(os.Stderr),
|
progressbar.OptionSetWriter(os.Stderr),
|
||||||
progressbar.OptionSetDescription(label),
|
progressbar.OptionSetDescription(label),
|
||||||
@@ -93,7 +81,9 @@ func newBar(label string, total int64) *progressbar.ProgressBar {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
return progressbar.NewOptions64(total, opts...)
|
p.bar = progressbar.NewOptions64(total, opts...)
|
||||||
|
|
||||||
|
return p
|
||||||
}
|
}
|
||||||
|
|
||||||
// increment records one completed item and refreshes the display.
|
// increment records one completed item and refreshes the display.
|
||||||
@@ -116,45 +106,26 @@ func (p *progress) increment() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// warnf prints a one-line warning to stderr without corrupting the bar.
|
// warnf prints a one-line warning to stderr without corrupting the bar.
|
||||||
// The whole message is escaped like a report's path columns, so a path
|
|
||||||
// holding a newline cannot split the warning.
|
|
||||||
func (p *progress) warnf(format string, args ...any) {
|
func (p *progress) warnf(format string, args ...any) {
|
||||||
if p == nil {
|
if p == nil {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
msg := escapePath(fmt.Sprintf(format, args...))
|
|
||||||
|
|
||||||
if p.bar != nil && p.total < 0 {
|
|
||||||
// The library also redraws a spinner from its own goroutine, so
|
|
||||||
// a direct write could land inside a redraw. The bar prints the
|
|
||||||
// warning itself, just before its next redraw.
|
|
||||||
_, _ = progressbar.Bprintln(p.bar, msg)
|
|
||||||
|
|
||||||
return
|
|
||||||
}
|
|
||||||
|
|
||||||
if p.bar != nil {
|
if p.bar != nil {
|
||||||
_ = p.bar.Clear()
|
_ = p.bar.Clear()
|
||||||
}
|
}
|
||||||
|
|
||||||
fmt.Fprintln(os.Stderr, msg)
|
fmt.Fprintf(os.Stderr, format+"\n", args...)
|
||||||
}
|
}
|
||||||
|
|
||||||
// finish terminates the pass's display. A bar whose pass stopped short
|
// finish terminates the pass's display.
|
||||||
// of its total, as an interrupted one does, is left as last drawn; the
|
|
||||||
// library's Finish would fill it up.
|
|
||||||
func (p *progress) finish() {
|
func (p *progress) finish() {
|
||||||
if p == nil {
|
if p == nil {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
if p.bar != nil {
|
if p.bar != nil {
|
||||||
if p.total >= 0 && p.count < p.total {
|
|
||||||
_ = p.bar.Exit()
|
|
||||||
} else {
|
|
||||||
_ = p.bar.Finish()
|
_ = p.bar.Finish()
|
||||||
}
|
|
||||||
|
|
||||||
fmt.Fprintln(os.Stderr)
|
fmt.Fprintln(os.Stderr)
|
||||||
|
|
||||||
|
|||||||
@@ -1,189 +0,0 @@
|
|||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
"path/filepath"
|
|
||||||
"strings"
|
|
||||||
"testing"
|
|
||||||
"time"
|
|
||||||
)
|
|
||||||
|
|
||||||
// spinnerIdle comfortably outlasts the 100ms interval at which the
|
|
||||||
// progressbar library redraws a spinner from its own goroutine.
|
|
||||||
const spinnerIdle = 500 * time.Millisecond
|
|
||||||
|
|
||||||
//nolint:paralleltest // replaces the process-wide os.Stderr
|
|
||||||
func TestStderrIsTTYFalseForNonTerminals(t *testing.T) {
|
|
||||||
r, pipe, err := os.Pipe()
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
regular, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
devNull, err := os.OpenFile(os.DevNull, os.O_WRONLY, 0)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
saved := os.Stderr
|
|
||||||
|
|
||||||
t.Cleanup(func() {
|
|
||||||
os.Stderr = saved
|
|
||||||
|
|
||||||
for _, f := range []*os.File{r, pipe, regular, devNull} {
|
|
||||||
_ = f.Close()
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
cases := map[string]*os.File{
|
|
||||||
"a pipe": pipe,
|
|
||||||
"a regular file": regular,
|
|
||||||
os.DevNull: devNull,
|
|
||||||
}
|
|
||||||
|
|
||||||
for name, f := range cases {
|
|
||||||
os.Stderr = f
|
|
||||||
|
|
||||||
if stderrIsTTY() {
|
|
||||||
t.Errorf("stderrIsTTY() = true with stderr on %s", name)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestNewProgressPrintsBeforeFirstItem checks that each pass shows its
|
|
||||||
// zero state the moment it starts when stderr is not a terminal, and
|
|
||||||
// that the next line still waits for plainInterval.
|
|
||||||
//
|
|
||||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
|
||||||
func TestNewProgressPrintsBeforeFirstItem(t *testing.T) {
|
|
||||||
stderr := captureStderr(t)
|
|
||||||
|
|
||||||
newProgress("walk", -1).increment()
|
|
||||||
newProgress("hash", 10).increment()
|
|
||||||
|
|
||||||
want := "walk: 0 files, elapsed 0s\n" +
|
|
||||||
"hash: [0/10] 0% 0 files/s elapsed 0s eta ?\n"
|
|
||||||
if got := stderr(); got != want {
|
|
||||||
t.Errorf("stderr = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// newWalkSpinner returns the walk pass's terminal display, writing to
|
|
||||||
// os.Stderr whether or not it is a terminal, and stops the library's
|
|
||||||
// redraws when the test ends.
|
|
||||||
func newWalkSpinner(t *testing.T) *progress {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
p := &progress{
|
|
||||||
label: "walk", total: -1, start: time.Now(),
|
|
||||||
bar: newBar("walk", -1),
|
|
||||||
}
|
|
||||||
t.Cleanup(p.finish)
|
|
||||||
|
|
||||||
return p
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestProgressWarningsOnOwnLines drives the terminal display of the walk
|
|
||||||
// pass through a run of warnings with no items between them, as when the
|
|
||||||
// walk meets many unreadable paths, for several of the spinner's
|
|
||||||
// redraws: every warning must land on a line of its own, never inside a
|
|
||||||
// redraw.
|
|
||||||
//
|
|
||||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
|
||||||
func TestProgressWarningsOnOwnLines(t *testing.T) {
|
|
||||||
stderr := captureStderr(t)
|
|
||||||
p := newWalkSpinner(t)
|
|
||||||
|
|
||||||
// No pause between warnings: one written straight to stderr is
|
|
||||||
// garbled only if a redraw lands while it is being written.
|
|
||||||
issued := 0
|
|
||||||
for start := time.Now(); time.Since(start) < spinnerIdle; issued++ {
|
|
||||||
p.warnf("warning")
|
|
||||||
}
|
|
||||||
|
|
||||||
// The spinner prints the warnings at its next redraw.
|
|
||||||
time.Sleep(spinnerIdle)
|
|
||||||
|
|
||||||
// A terminal shows each line as the text after its last carriage
|
|
||||||
// return.
|
|
||||||
shown := 0
|
|
||||||
|
|
||||||
for line := range strings.SplitSeq(stderr(), "\n") {
|
|
||||||
if !strings.Contains(line, "warning") {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
shown++
|
|
||||||
|
|
||||||
if text := line[strings.LastIndex(line, "\r")+1:]; text != "warning" {
|
|
||||||
t.Errorf("terminal shows %q, want %q", text, "warning")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if shown != issued {
|
|
||||||
t.Errorf("%d warning lines, want %d", shown, issued)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestSpinnerShowsCountAfterBurst checks that once a burst of items
|
|
||||||
// faster than the redraw limit is over, the walk display shows every
|
|
||||||
// item completed while it waits for the next one.
|
|
||||||
//
|
|
||||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
|
||||||
func TestSpinnerShowsCountAfterBurst(t *testing.T) {
|
|
||||||
stderr := captureStderr(t)
|
|
||||||
p := newWalkSpinner(t)
|
|
||||||
|
|
||||||
for range 50 {
|
|
||||||
p.increment()
|
|
||||||
}
|
|
||||||
|
|
||||||
time.Sleep(spinnerIdle)
|
|
||||||
|
|
||||||
if shown := lastFrame(stderr()); !strings.Contains(shown, "(50/-,") {
|
|
||||||
t.Errorf("terminal shows %q, want a count of 50", shown)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// lastFrame returns what a terminal shows of the frames a bar drew: the
|
|
||||||
// last one. The library starts each frame with a carriage return and
|
|
||||||
// erases the previous one with spaces first.
|
|
||||||
func lastFrame(out string) string {
|
|
||||||
var shown string
|
|
||||||
|
|
||||||
for frame := range strings.SplitSeq(out, "\r") {
|
|
||||||
if strings.TrimSpace(frame) != "" {
|
|
||||||
shown = frame
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return shown
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestBarStoppedShortKeepsCount checks that the terminal display of a
|
|
||||||
// pass that stops before its total, as an interrupted one does, is left
|
|
||||||
// as last drawn instead of being filled up.
|
|
||||||
//
|
|
||||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
|
||||||
func TestBarStoppedShortKeepsCount(t *testing.T) {
|
|
||||||
stderr := captureStderr(t)
|
|
||||||
p := &progress{
|
|
||||||
label: "hash", total: 10, start: time.Now(),
|
|
||||||
bar: newBar("hash", 10),
|
|
||||||
}
|
|
||||||
|
|
||||||
p.increment()
|
|
||||||
|
|
||||||
// Past the redraw limit, so the bar draws the next count.
|
|
||||||
time.Sleep(2 * barThrottle)
|
|
||||||
p.increment()
|
|
||||||
p.finish()
|
|
||||||
|
|
||||||
if shown := lastFrame(stderr()); !strings.Contains(shown, "(2/10,") {
|
|
||||||
t.Errorf("terminal shows %q, want a count of 2 of 10", shown)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -4,10 +4,9 @@ import (
|
|||||||
"bufio"
|
"bufio"
|
||||||
"context"
|
"context"
|
||||||
"fmt"
|
"fmt"
|
||||||
"io"
|
|
||||||
"os"
|
"os"
|
||||||
|
"slices"
|
||||||
"strings"
|
"strings"
|
||||||
"time"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// ioBufSize is the buffer size for the buffered stdout writers.
|
// ioBufSize is the buffer size for the buffered stdout writers.
|
||||||
@@ -22,66 +21,80 @@ const minGroupSize = 2
|
|||||||
// only and used by scan for change detection.
|
// only and used by scan for change detection.
|
||||||
type scanRec struct {
|
type scanRec struct {
|
||||||
size int64
|
size int64
|
||||||
mtime time.Time
|
mtime int64
|
||||||
head string
|
head string
|
||||||
tail string
|
tail string
|
||||||
content string
|
content string
|
||||||
path string
|
path string
|
||||||
}
|
}
|
||||||
|
|
||||||
// runReport implements the report subcommand: it prints the file-level
|
// loadRecords opens the database and reads every file record for the
|
||||||
// duplicates report as TSV on stdout. SQLite groups and orders the
|
// report and trees subcommands. Any database problem — including a
|
||||||
// records, and each row is written as it is read, so no group is held
|
// missing database — is fatal. The error is returned rather than
|
||||||
// in memory. It never touches the scanned filesystem; its only I/O is
|
// exiting, so that the deferred close — which checkpoints the SQLite
|
||||||
// the database (with SQLite's temporary sort file), stdout, and stderr.
|
// WAL — always runs; the database is closed before the caller formats
|
||||||
// Any database problem, including a missing database, is fatal.
|
// its output, so it stays closed even if that output fails.
|
||||||
func runReport(ctx context.Context, stdout io.Writer) error {
|
func loadRecords(ctx context.Context) ([]scanRec, error) {
|
||||||
dbPath := databasePath()
|
dbPath := databasePath()
|
||||||
|
|
||||||
db, err := openReportDatabase(ctx, dbPath)
|
db, err := openReportDatabase(ctx, dbPath)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
defer func() { _ = db.Close() }()
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
out := bufio.NewWriterSize(stdout, ioBufSize)
|
recs, err := loadFileRows(ctx, db)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("database %s: %w", dbPath, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return recs, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// dupeGroup is one set of candidate-duplicate files: identical size,
|
||||||
|
// head hash, tail hash, and content hash. paths is sorted
|
||||||
|
// lexicographically; the first entry is the group's "first", the rest
|
||||||
|
// are dupes.
|
||||||
|
type dupeGroup struct {
|
||||||
|
size int64
|
||||||
|
paths []string
|
||||||
|
}
|
||||||
|
|
||||||
|
// runReport implements the report subcommand: it reads every record
|
||||||
|
// from the database and prints the file-level duplicates report as TSV
|
||||||
|
// on stdout. It never touches the scanned filesystem; its only I/O is
|
||||||
|
// the database, stdout, and stderr.
|
||||||
|
func runReport(ctx context.Context) error {
|
||||||
|
recs, err := loadRecords(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
dupes := collectDupeGroups(recs)
|
||||||
|
|
||||||
|
out := bufio.NewWriterSize(os.Stdout, ioBufSize)
|
||||||
|
|
||||||
_, err = fmt.Fprintln(out, "first\tdupe\tsize")
|
_, err = fmt.Fprintln(out, "first\tdupe\tsize")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("write stdout: %w", err)
|
return fmt.Errorf("write stdout: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
var (
|
dupeFiles := 0
|
||||||
groups, dupeFiles int
|
|
||||||
reclaimable int64
|
|
||||||
writeErr error
|
|
||||||
)
|
|
||||||
|
|
||||||
records, err := loadDupeRows(ctx, db,
|
var reclaimable int64
|
||||||
func(first, path string, size int64) error {
|
|
||||||
// A group's first path is its first row; every other
|
|
||||||
// path is a dupe.
|
|
||||||
if path == first {
|
|
||||||
groups++
|
|
||||||
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
_, writeErr = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
|
||||||
escapePath(first), escapePath(path), size)
|
|
||||||
dupeFiles++
|
|
||||||
reclaimable += size
|
|
||||||
|
|
||||||
return writeErr
|
|
||||||
})
|
|
||||||
|
|
||||||
if writeErr != nil {
|
|
||||||
return fmt.Errorf("write stdout: %w", writeErr)
|
|
||||||
}
|
|
||||||
|
|
||||||
|
for _, g := range dupes {
|
||||||
|
for _, p := range g.paths[1:] {
|
||||||
|
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
||||||
|
g.paths[0], p, g.size)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("database %s: %w", dbPath, err)
|
return fmt.Errorf("write stdout: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
dupeFiles++
|
||||||
|
reclaimable += g.size
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
err = out.Flush()
|
err = out.Flush()
|
||||||
@@ -92,24 +105,55 @@ func runReport(ctx context.Context, stdout io.Writer) error {
|
|||||||
fmt.Fprintf(os.Stderr,
|
fmt.Fprintf(os.Stderr,
|
||||||
"report: %d records read, %d duplicate groups, %d dupe files, "+
|
"report: %d records read, %d duplicate groups, %d dupe files, "+
|
||||||
"%s reclaimable\n",
|
"%s reclaimable\n",
|
||||||
records, groups, dupeFiles, humanBytes(reclaimable))
|
len(recs), len(dupes), dupeFiles, humanBytes(reclaimable))
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// escapePath returns a path as it is written in a report column (README
|
// collectDupeGroups groups records by signature and returns every group
|
||||||
// "Report output format"): a backslash, tab, newline or carriage return
|
// with two or more paths, each group's paths sorted lexicographically,
|
||||||
// becomes \\, \t, \n or \r, and every other byte is kept as it is.
|
// groups ordered by size descending then by first path ascending.
|
||||||
// Grouping and sorting use the raw path, never this form.
|
func collectDupeGroups(recs []scanRec) []dupeGroup {
|
||||||
func escapePath(p string) string {
|
groups := make(map[fileSig][]string)
|
||||||
// Most paths need no escaping; skip building a replacer for them.
|
|
||||||
if !strings.ContainsAny(p, "\\\t\n\r") {
|
for _, r := range recs {
|
||||||
return p
|
// A record without a content hash has unknown content and is
|
||||||
|
// never reported as a duplicate (README "Database").
|
||||||
|
if r.content == "" {
|
||||||
|
continue
|
||||||
}
|
}
|
||||||
|
|
||||||
return strings.NewReplacer(
|
k := fileSig{
|
||||||
`\`, `\\`, "\t", `\t`, "\n", `\n`, "\r", `\r`,
|
size: r.size, head: r.head, tail: r.tail, content: r.content,
|
||||||
).Replace(p)
|
}
|
||||||
|
groups[k] = append(groups[k], r.path)
|
||||||
|
}
|
||||||
|
|
||||||
|
var dupes []dupeGroup
|
||||||
|
|
||||||
|
for k, paths := range groups {
|
||||||
|
if len(paths) < minGroupSize {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
slices.Sort(paths)
|
||||||
|
dupes = append(dupes, dupeGroup{size: k.size, paths: paths})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Biggest reclaimable space first; ties broken by first path.
|
||||||
|
slices.SortFunc(dupes, func(a, b dupeGroup) int {
|
||||||
|
if a.size != b.size {
|
||||||
|
if a.size > b.size {
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
return strings.Compare(a.paths[0], b.paths[0])
|
||||||
|
})
|
||||||
|
|
||||||
|
return dupes
|
||||||
}
|
}
|
||||||
|
|
||||||
// humanBytes formats a byte count in human units (binary prefixes).
|
// humanBytes formats a byte count in human units (binary prefixes).
|
||||||
|
|||||||
+17
-269
@@ -1,255 +1,11 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
|
||||||
"database/sql"
|
|
||||||
"errors"
|
|
||||||
"fmt"
|
|
||||||
"io"
|
|
||||||
"os"
|
|
||||||
"path/filepath"
|
|
||||||
"slices"
|
"slices"
|
||||||
"strings"
|
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// awkwardDir is a directory name holding every byte the reports escape.
|
func TestCollectDupeGroups(t *testing.T) {
|
||||||
const awkwardDir = "/d/\tone\ntwo\rthree\\four"
|
|
||||||
|
|
||||||
// awkwardPairRecs is a duplicate pair in sibling directories /d/A and
|
|
||||||
// awkwardDir. A raw tab sorts before "A" but its escaped form `\t`
|
|
||||||
// sorts after it, so awkwardDir coming first shows that sorting uses
|
|
||||||
// the raw path.
|
|
||||||
func awkwardPairRecs() []scanRec {
|
|
||||||
return []scanRec{
|
|
||||||
{size: 5, head: "h", tail: "t", content: "c", path: "/d/A/f"},
|
|
||||||
{size: 5, head: "h", tail: "t", content: "c", path: awkwardDir + "/f"},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// seedDatabase writes recs into a fresh database and returns its path.
|
|
||||||
func seedDatabase(t *testing.T, recs []scanRec) string {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
path := testDBPath(t)
|
|
||||||
|
|
||||||
db, err := openScanDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = applyChanges(t.Context(), db, recs, nil, nil)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = db.Close()
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return path
|
|
||||||
}
|
|
||||||
|
|
||||||
// dupeGroup is one duplicate group as report reads it: the size, and
|
|
||||||
// the paths in report order, first path first.
|
|
||||||
type dupeGroup struct {
|
|
||||||
size int64
|
|
||||||
paths []string
|
|
||||||
}
|
|
||||||
|
|
||||||
// dupeGroups returns the duplicate groups report reads from db, in
|
|
||||||
// report order.
|
|
||||||
func dupeGroups(t *testing.T, db *sql.DB) []dupeGroup {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
var groups []dupeGroup
|
|
||||||
|
|
||||||
_, err := loadDupeRows(t.Context(), db,
|
|
||||||
func(first, path string, size int64) error {
|
|
||||||
if path == first {
|
|
||||||
groups = append(groups, dupeGroup{size: size})
|
|
||||||
}
|
|
||||||
|
|
||||||
g := &groups[len(groups)-1]
|
|
||||||
g.paths = append(g.paths, path)
|
|
||||||
|
|
||||||
return nil
|
|
||||||
})
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return groups
|
|
||||||
}
|
|
||||||
|
|
||||||
// dupeGroupsOf writes recs into a fresh database and returns the
|
|
||||||
// duplicate groups report reads from it.
|
|
||||||
func dupeGroupsOf(t *testing.T, recs []scanRec) []dupeGroup {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
db := openTestDB(t)
|
|
||||||
|
|
||||||
err := applyChanges(t.Context(), db, recs, nil, nil)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return dupeGroups(t, db)
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunReportEscapesPaths(t *testing.T) {
|
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{cmdReport}, &stdout, &stderr)
|
|
||||||
if code != exitOK {
|
|
||||||
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
|
||||||
code, exitOK, stderr.String())
|
|
||||||
}
|
|
||||||
|
|
||||||
want := "first\tdupe\tsize\n" +
|
|
||||||
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
|
|
||||||
if got := stdout.String(); got != want {
|
|
||||||
t.Errorf("stdout = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestReportStdoutFailsWhileReading(t *testing.T) {
|
|
||||||
// Each row holds two paths longer than dir, so the report is more
|
|
||||||
// than twice the stdout buffer and stdout fails while rows are
|
|
||||||
// still being read, not at the final flush.
|
|
||||||
dir := "/" + strings.Repeat("d", 4096)
|
|
||||||
|
|
||||||
recs := make([]scanRec, ioBufSize/len(dir))
|
|
||||||
for i := range recs {
|
|
||||||
recs[i] = scanRec{
|
|
||||||
size: 1, head: "h", tail: "t", content: "c",
|
|
||||||
path: fmt.Sprintf("%s/%d", dir, i),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, recs))
|
|
||||||
|
|
||||||
err := runReport(t.Context(), failingWriter{})
|
|
||||||
if !errors.Is(err, errWriteFailed) ||
|
|
||||||
!strings.HasPrefix(err.Error(), "write stdout: ") {
|
|
||||||
t.Errorf("error = %v, want write stdout: %v", err, errWriteFailed)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunReportsIgnoreInsertionOrder(t *testing.T) {
|
|
||||||
// README §Constraints: identical database contents give identical
|
|
||||||
// output, whatever order the records were inserted in.
|
|
||||||
recs := append(smokeTreeRecs(), awkwardPairRecs()...)
|
|
||||||
recs = append(recs,
|
|
||||||
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/2"},
|
|
||||||
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/1"},
|
|
||||||
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/2"},
|
|
||||||
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/1"},
|
|
||||||
scanRec{size: 50, path: "/x/unhashed"},
|
|
||||||
)
|
|
||||||
|
|
||||||
reversed := slices.Clone(recs)
|
|
||||||
slices.Reverse(reversed)
|
|
||||||
|
|
||||||
for _, name := range []string{cmdReport, cmdTrees} {
|
|
||||||
t.Run(name, func(t *testing.T) {
|
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, recs))
|
|
||||||
|
|
||||||
forward := runStdout(t, name)
|
|
||||||
|
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, reversed))
|
|
||||||
|
|
||||||
backward := runStdout(t, name)
|
|
||||||
|
|
||||||
if strings.Count(forward, "\n") < 3 {
|
|
||||||
t.Errorf("stdout = %q, want at least two rows", forward)
|
|
||||||
}
|
|
||||||
|
|
||||||
if forward != backward {
|
|
||||||
t.Errorf("stdout depends on insertion order: %q vs %q",
|
|
||||||
forward, backward)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// runStdout runs the subcommand name and returns its stdout, failing
|
|
||||||
// the test unless it succeeds.
|
|
||||||
func runStdout(t *testing.T, name string) string {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{name}, &stdout, &stderr)
|
|
||||||
if code != exitOK {
|
|
||||||
t.Fatalf("run(%s) = %d, want %d; stderr: %s",
|
|
||||||
name, code, exitOK, stderr.String())
|
|
||||||
}
|
|
||||||
|
|
||||||
return stdout.String()
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestEscapePath(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
cases := map[string]string{
|
|
||||||
"/srv/plain": "/srv/plain",
|
|
||||||
"/a\tb": `/a\tb`,
|
|
||||||
"/a\nb": `/a\nb`,
|
|
||||||
"/a\rb": `/a\rb`,
|
|
||||||
`/a\b`: `/a\\b`,
|
|
||||||
`/a\tb`: `/a\\tb`,
|
|
||||||
"/not-utf8\xff": "/not-utf8\xff",
|
|
||||||
}
|
|
||||||
for in, want := range cases {
|
|
||||||
if got := escapePath(in); got != want {
|
|
||||||
t.Errorf("escapePath(%q) = %q, want %q", in, got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestWarnfEscapes checks that a warning naming a path that holds a
|
|
||||||
// newline is still one line.
|
|
||||||
//
|
|
||||||
//nolint:paralleltest // replaces the process-wide os.Stderr
|
|
||||||
func TestWarnfEscapes(t *testing.T) {
|
|
||||||
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
saved := os.Stderr
|
|
||||||
os.Stderr = f
|
|
||||||
|
|
||||||
t.Cleanup(func() {
|
|
||||||
os.Stderr = saved
|
|
||||||
|
|
||||||
_ = f.Close()
|
|
||||||
})
|
|
||||||
|
|
||||||
(&progress{}).warnf("stat %s: %s", "/d/a\nb", "gone")
|
|
||||||
|
|
||||||
_, err = f.Seek(0, io.SeekStart)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
got, err := io.ReadAll(f)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
want := `stat /d/a\nb: gone` + "\n"
|
|
||||||
if string(got) != want {
|
|
||||||
t.Errorf("warning = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestDupeGroups(t *testing.T) {
|
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
@@ -264,7 +20,7 @@ func TestDupeGroups(t *testing.T) {
|
|||||||
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := dupeGroupsOf(t, recs)
|
groups := collectDupeGroups(recs)
|
||||||
if len(groups) != 2 {
|
if len(groups) != 2 {
|
||||||
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
||||||
}
|
}
|
||||||
@@ -282,7 +38,7 @@ func TestDupeGroups(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestDupeGroupsContentSeparates(t *testing.T) {
|
func TestCollectDupeGroupsContentSeparates(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// Same size, head, and tail, but different content hashes: the final
|
// Same size, head, and tail, but different content hashes: the final
|
||||||
@@ -297,7 +53,7 @@ func TestDupeGroupsContentSeparates(t *testing.T) {
|
|||||||
{size: 100, head: "h", tail: "t", path: "/e"},
|
{size: 100, head: "h", tail: "t", path: "/e"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := dupeGroupsOf(t, recs)
|
groups := collectDupeGroups(recs)
|
||||||
if len(groups) != 1 {
|
if len(groups) != 1 {
|
||||||
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
||||||
len(groups))
|
len(groups))
|
||||||
@@ -308,41 +64,33 @@ func TestDupeGroupsContentSeparates(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestDupeGroupsMtimeExcluded(t *testing.T) {
|
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// mtime is informational only; records differing only in mtime
|
// mtime is informational only; records differing only in mtime
|
||||||
// still group together.
|
// still group together.
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{
|
{size: 9, mtime: 100, head: "h", tail: "t", content: "c", path: "/m/1"},
|
||||||
size: 9, mtime: time.Unix(100, 0), head: "h", tail: "t",
|
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
||||||
content: "c", path: "/m/1",
|
|
||||||
},
|
|
||||||
{
|
|
||||||
size: 9, mtime: time.Unix(200, 0), head: "h", tail: "t",
|
|
||||||
content: "c", path: "/m/2",
|
|
||||||
},
|
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := dupeGroupsOf(t, recs)
|
groups := collectDupeGroups(recs)
|
||||||
if len(groups) != 1 {
|
if len(groups) != 1 {
|
||||||
t.Fatalf("len(groups) = %d, want 1", len(groups))
|
t.Fatalf("len(groups) = %d, want 1", len(groups))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestDupeGroupsTieBreak(t *testing.T) {
|
func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// The hashes sort opposite to the first paths, so ordering the
|
|
||||||
// groups by hash instead of by first path fails this test.
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/2"},
|
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"},
|
||||||
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/1"},
|
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"},
|
||||||
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/2"},
|
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"},
|
||||||
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/1"},
|
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := dupeGroupsOf(t, recs)
|
groups := collectDupeGroups(recs)
|
||||||
if len(groups) != 2 {
|
if len(groups) != 2 {
|
||||||
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
||||||
}
|
}
|
||||||
@@ -354,7 +102,7 @@ func TestDupeGroupsTieBreak(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestDupeGroupsDeterministic(t *testing.T) {
|
func TestCollectDupeGroupsDeterministic(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
@@ -364,12 +112,12 @@ func TestDupeGroupsDeterministic(t *testing.T) {
|
|||||||
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
||||||
}
|
}
|
||||||
|
|
||||||
forward := dupeGroupsOf(t, recs)
|
forward := collectDupeGroups(recs)
|
||||||
|
|
||||||
reversed := slices.Clone(recs)
|
reversed := slices.Clone(recs)
|
||||||
slices.Reverse(reversed)
|
slices.Reverse(reversed)
|
||||||
|
|
||||||
backward := dupeGroupsOf(t, reversed)
|
backward := collectDupeGroups(reversed)
|
||||||
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
|
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
|
||||||
return a.size == b.size && slices.Equal(a.paths, b.paths)
|
return a.size == b.size && slices.Equal(a.paths, b.paths)
|
||||||
}) {
|
}) {
|
||||||
|
|||||||
@@ -11,13 +11,11 @@ import (
|
|||||||
"io"
|
"io"
|
||||||
"io/fs"
|
"io/fs"
|
||||||
"os"
|
"os"
|
||||||
"os/signal"
|
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"slices"
|
"slices"
|
||||||
"strings"
|
"strings"
|
||||||
"sync"
|
"sync"
|
||||||
"syscall"
|
"syscall"
|
||||||
"time"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// The duplicate ladder (see hashSignature and README "Duplicate
|
// The duplicate ladder (see hashSignature and README "Duplicate
|
||||||
@@ -59,17 +57,13 @@ const sampleWindow = 1024 * 1024
|
|||||||
// and hash worker pools.
|
// and hash worker pools.
|
||||||
const workQueueDepth = 1024
|
const workQueueDepth = 1024
|
||||||
|
|
||||||
// errInterrupted reports a scan stopped by SIGINT or SIGTERM. runScan
|
|
||||||
// has already printed its line, so run prints nothing more.
|
|
||||||
var errInterrupted = errors.New("scan interrupted")
|
|
||||||
|
|
||||||
// fileRec carries one statted file between the scan phases. dev and
|
// fileRec carries one statted file between the scan phases. dev and
|
||||||
// ino identify the underlying inode so hard-linked paths can share
|
// ino identify the underlying inode so hard-linked paths can share
|
||||||
// one read; both are zero when the platform exposes no inode.
|
// one read; both are zero when the platform exposes no inode.
|
||||||
type fileRec struct {
|
type fileRec struct {
|
||||||
path string
|
path string
|
||||||
size int64
|
size int64
|
||||||
mtime time.Time
|
mtime int64
|
||||||
dev uint64
|
dev uint64
|
||||||
ino uint64
|
ino uint64
|
||||||
}
|
}
|
||||||
@@ -80,7 +74,7 @@ type fileRec struct {
|
|||||||
// they would dominate the scan's memory.
|
// they would dominate the scan's memory.
|
||||||
type fileMeta struct {
|
type fileMeta struct {
|
||||||
size int64
|
size int64
|
||||||
mtime time.Time
|
mtime int64
|
||||||
hashed bool
|
hashed bool
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -91,19 +85,17 @@ type fileMeta struct {
|
|||||||
// least one other file shares are ever hashed: a size-unique file
|
// least one other file shares are ever hashed: a size-unique file
|
||||||
// cannot be a duplicate. A file of headTailMin or more gets its content
|
// cannot be a duplicate. A file of headTailMin or more gets its content
|
||||||
// hash only when its size, head, and tail match another file's. Flag
|
// hash only when its size, head, and tail match another file's. Flag
|
||||||
// parsing and the at-least-one-operand check are done by cobra. The
|
// parsing and the at-least-one-operand check are done by cobra. Errors
|
||||||
// scan holds the lock on the database for its whole run, so a second
|
// are returned rather than exiting, so that the deferred close — which
|
||||||
// scan fails before it walks the filesystem or opens the database.
|
// checkpoints the SQLite WAL — always runs. Cancelling ctx unwinds the
|
||||||
// Errors are returned rather than exiting, so that the deferred close —
|
// worker pools and aborts the scan with the context's error.
|
||||||
// which takes the database out of WAL mode — always runs, and the lock
|
|
||||||
// is released after it. When ctx is cancelled, as by the SIGINT or
|
|
||||||
// SIGTERM that interruptContext catches, the scan keeps what it has
|
|
||||||
// hashed (see syncScan), prints how many files its walk reached, and
|
|
||||||
// returns errInterrupted. workers must be at least 1; the scan command
|
|
||||||
// rejects anything less.
|
|
||||||
func runScan(ctx context.Context, roots []string, workers int,
|
func runScan(ctx context.Context, roots []string, workers int,
|
||||||
oneFS bool,
|
oneFS bool,
|
||||||
) error {
|
) error {
|
||||||
|
if workers < 1 {
|
||||||
|
workers = 1
|
||||||
|
}
|
||||||
|
|
||||||
roots, err := resolveRoots(roots)
|
roots, err := resolveRoots(roots)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
@@ -111,31 +103,14 @@ func runScan(ctx context.Context, roots []string, workers int,
|
|||||||
|
|
||||||
dbPath := databasePath()
|
dbPath := databasePath()
|
||||||
|
|
||||||
lock, err := lockScanDatabase(dbPath)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = lock.Close() }()
|
|
||||||
|
|
||||||
db, err := openScanDatabase(ctx, dbPath)
|
db, err := openScanDatabase(ctx, dbPath)
|
||||||
if err != nil && ctx.Err() != nil {
|
|
||||||
// Interrupted while opening; SQLite may report that with an
|
|
||||||
// error of its own rather than the context's.
|
|
||||||
return interrupted(0)
|
|
||||||
}
|
|
||||||
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
defer closeScanDatabase(ctx, db, dbPath)
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
st, err := syncScan(ctx, db, roots, workers, oneFS)
|
st, err := syncScan(ctx, db, roots, workers, oneFS)
|
||||||
if errors.Is(err, context.Canceled) {
|
|
||||||
return interrupted(st.walked)
|
|
||||||
}
|
|
||||||
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("update database %s: %w", dbPath, err)
|
return fmt.Errorf("update database %s: %w", dbPath, err)
|
||||||
}
|
}
|
||||||
@@ -149,34 +124,6 @@ func runScan(ctx context.Context, roots []string, workers int,
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// interruptContext returns a copy of ctx that the first SIGINT or
|
|
||||||
// SIGTERM cancels; the scan command runs the scan under it. stop
|
|
||||||
// releases the signals.
|
|
||||||
func interruptContext(ctx context.Context) (context.Context, func()) {
|
|
||||||
// A SIGINT ignored from the start, as by a script's background job,
|
|
||||||
// stays ignored.
|
|
||||||
signals := []os.Signal{syscall.SIGTERM}
|
|
||||||
if !signal.Ignored(syscall.SIGINT) {
|
|
||||||
signals = append(signals, syscall.SIGINT)
|
|
||||||
}
|
|
||||||
|
|
||||||
ctx, stop := signal.NotifyContext(ctx, signals...)
|
|
||||||
|
|
||||||
// Stopping restores the default handling, so a second signal ends
|
|
||||||
// the process at once.
|
|
||||||
context.AfterFunc(ctx, stop)
|
|
||||||
|
|
||||||
return ctx, stop
|
|
||||||
}
|
|
||||||
|
|
||||||
// interrupted prints the line for a scan stopped by a signal after its
|
|
||||||
// walk reached walked files, and returns errInterrupted.
|
|
||||||
func interrupted(walked int) error {
|
|
||||||
fmt.Fprintf(os.Stderr, "scan: interrupted after %d files\n", walked)
|
|
||||||
|
|
||||||
return errInterrupted
|
|
||||||
}
|
|
||||||
|
|
||||||
// resolveRoots converts each PATH operand to an absolute, lexically
|
// resolveRoots converts each PATH operand to an absolute, lexically
|
||||||
// cleaned path (symlinks are not resolved) and verifies that it
|
// cleaned path (symlinks are not resolved) and verifies that it
|
||||||
// exists. Database records are keyed by absolute path, so scan results
|
// exists. Database records are keyed by absolute path, so scan results
|
||||||
@@ -230,9 +177,8 @@ func pruneRoots(roots []string) []string {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// scanStats summarizes one scan's database synchronization for the
|
// scanStats summarizes one scan's database synchronization for the
|
||||||
// final stderr summary, or for the line an interrupted scan prints.
|
// final stderr summary.
|
||||||
type scanStats struct {
|
type scanStats struct {
|
||||||
walked int // files the walk reached
|
|
||||||
added int
|
added int
|
||||||
updated int
|
updated int
|
||||||
removed int
|
removed int
|
||||||
@@ -254,30 +200,7 @@ type scanState struct {
|
|||||||
st scanStats
|
st scanStats
|
||||||
}
|
}
|
||||||
|
|
||||||
// syncScan synchronizes the database with the filesystem under roots;
|
// syncScan synchronizes the database with the filesystem under roots
|
||||||
// see runPhases. When ctx is cancelled, as by an interrupt, it commits
|
|
||||||
// the hashed records still waiting in the batch, starts no other write
|
|
||||||
// or deletion, and returns the cancellation.
|
|
||||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
|
||||||
workers int, oneFS bool,
|
|
||||||
) (scanStats, error) {
|
|
||||||
s := &scanState{db: db}
|
|
||||||
|
|
||||||
err := s.runPhases(ctx, roots, workers, oneFS)
|
|
||||||
if err == nil || ctx.Err() == nil {
|
|
||||||
return s.st, err
|
|
||||||
}
|
|
||||||
|
|
||||||
// The one write made after the cancellation, so it cannot use ctx.
|
|
||||||
err = applyChanges(context.WithoutCancel(ctx), db, s.batch, nil, nil)
|
|
||||||
if err != nil {
|
|
||||||
return s.st, err
|
|
||||||
}
|
|
||||||
|
|
||||||
return s.st, ctx.Err()
|
|
||||||
}
|
|
||||||
|
|
||||||
// runPhases synchronizes the database with the filesystem under roots
|
|
||||||
// in four sequential phases: walk (enumerate and stat every file,
|
// in four sequential phases: walk (enumerate and stat every file,
|
||||||
// building a complete size census), hash (read only the new or
|
// building a complete size census), hash (read only the new or
|
||||||
// changed — or previously unhashed — files whose size at least one
|
// changed — or previously unhashed — files whose size at least one
|
||||||
@@ -287,73 +210,47 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
|||||||
// in the content hash of every record of headTailMin or more whose
|
// in the content hash of every record of headTailMin or more whose
|
||||||
// size, head, and tail match another record's). Records outside the
|
// size, head, and tail match another record's). Records outside the
|
||||||
// roots are never touched, except that the content phase fills in
|
// roots are never touched, except that the content phase fills in
|
||||||
// their content hash. Operands the walk cannot start from are dropped
|
// their content hash.
|
||||||
// first, so the records beneath them count as outside the roots unless
|
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||||
// they lie under another root.
|
|
||||||
func (s *scanState) runPhases(ctx context.Context, roots []string,
|
|
||||||
workers int, oneFS bool,
|
workers int, oneFS bool,
|
||||||
) error {
|
) (scanStats, error) {
|
||||||
// Types are checked before pruning so that an operand under a
|
roots = pruneRoots(roots)
|
||||||
// dropped one is still scanned, not dropped as lying under it.
|
|
||||||
roots = pruneRoots(s.walkableRoots(roots))
|
s := &scanState{db: db}
|
||||||
|
|
||||||
err := s.loadIndex(ctx, roots)
|
err := s.loadIndex(ctx, roots)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return s.st, err
|
||||||
}
|
}
|
||||||
|
|
||||||
changed, unhashed := s.walkPhase(startWalk(ctx, roots, oneFS, workers))
|
changed, unhashed := s.walkPhase(startWalk(ctx, roots, oneFS, workers))
|
||||||
|
|
||||||
// A cancelled walk stops early, so its size census covers only part
|
// A cancelled walk stops early, so its size census covers only part
|
||||||
// of the roots, and every file it never reached would look vanished
|
// of the roots, and every file it never reached looks vanished to
|
||||||
// to the update phase. Stop before anything is written or deleted.
|
// the update phase. Defence in depth rather than the only barrier:
|
||||||
|
// that phase would today fail on its first BeginTx with the same
|
||||||
|
// cancelled context before deleting anything. But it is the barrier
|
||||||
|
// that survives a later decision to let an interrupted scan commit
|
||||||
|
// what it has, and it turns a confusing failure deep in the update
|
||||||
|
// phase into a clean abort at the phase boundary.
|
||||||
err = ctx.Err()
|
err = ctx.Err()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return s.st, err
|
||||||
}
|
}
|
||||||
|
|
||||||
s.partition(changed, unhashed)
|
s.partition(changed, unhashed)
|
||||||
|
|
||||||
err = s.hashPhase(ctx, workers)
|
err = s.hashPhase(ctx, workers)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return s.st, err
|
||||||
}
|
}
|
||||||
|
|
||||||
err = s.updatePhase(ctx)
|
err = s.updatePhase(ctx)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return s.st, err
|
||||||
}
|
}
|
||||||
|
|
||||||
return s.contentPhase(ctx, workers)
|
return s.st, s.contentPhase(ctx, workers)
|
||||||
}
|
|
||||||
|
|
||||||
// walkableRoots returns the operands the walk can start from: regular
|
|
||||||
// files, and directories not named .zfs. Every other operand is warned
|
|
||||||
// about, counted as skipped, and dropped. A dropped operand is no
|
|
||||||
// longer a root, so the records stored beneath it count as outside the
|
|
||||||
// roots and are not deleted as unverified, unless it lies under another
|
|
||||||
// root. An operand that fails lstat here is kept, and the walk warns
|
|
||||||
// about it.
|
|
||||||
func (s *scanState) walkableRoots(roots []string) []string {
|
|
||||||
kept := make([]string, 0, len(roots))
|
|
||||||
|
|
||||||
for _, root := range roots {
|
|
||||||
fi, err := os.Lstat(root)
|
|
||||||
if err == nil {
|
|
||||||
warn := operandWarning(root, fi)
|
|
||||||
if warn != "" {
|
|
||||||
s.st.skipped++
|
|
||||||
|
|
||||||
fmt.Fprintln(os.Stderr, escapePath(warn))
|
|
||||||
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
kept = append(kept, root)
|
|
||||||
}
|
|
||||||
|
|
||||||
return kept
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// loadIndex indexes the database records under the scan roots for
|
// loadIndex indexes the database records under the scan roots for
|
||||||
@@ -370,7 +267,7 @@ func (s *scanState) loadIndex(ctx context.Context, roots []string) error {
|
|||||||
s.existing = make(map[string]fileMeta)
|
s.existing = make(map[string]fileMeta)
|
||||||
|
|
||||||
return loadFileMeta(ctx, s.db,
|
return loadFileMeta(ctx, s.db,
|
||||||
func(path string, size int64, mtime time.Time, hashed bool) {
|
func(path string, size, mtime int64, hashed bool) {
|
||||||
prog.increment()
|
prog.increment()
|
||||||
|
|
||||||
if underAnyRoot(path, roots) {
|
if underAnyRoot(path, roots) {
|
||||||
@@ -409,14 +306,13 @@ func (s *scanState) walkPhase(
|
|||||||
}
|
}
|
||||||
|
|
||||||
s.sizes = append(s.sizes, ev.rec.size)
|
s.sizes = append(s.sizes, ev.rec.size)
|
||||||
s.st.walked++
|
|
||||||
|
|
||||||
prog.increment()
|
prog.increment()
|
||||||
|
|
||||||
old, ok := s.existing[ev.rec.path]
|
old, ok := s.existing[ev.rec.path]
|
||||||
|
|
||||||
switch {
|
switch {
|
||||||
case !ok || old.size != ev.rec.size || mtimeAfter(ev.rec.mtime, old.mtime):
|
case !ok || old.size != ev.rec.size || old.mtime < ev.rec.mtime:
|
||||||
changed = append(changed, ev.rec)
|
changed = append(changed, ev.rec)
|
||||||
case old.hashed:
|
case old.hashed:
|
||||||
delete(s.existing, ev.rec.path)
|
delete(s.existing, ev.rec.path)
|
||||||
@@ -613,22 +509,16 @@ func (s *scanState) recordRun(ctx context.Context, r hashResult) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// commitFullBatch commits the running batch once it holds
|
// commitFullBatch commits the running batch once it holds
|
||||||
// updateBatchSize records. A batch that fails to commit is kept: the
|
// updateBatchSize records.
|
||||||
// commit fails when the scan is interrupted, and syncScan then commits
|
|
||||||
// the batch itself.
|
|
||||||
func (s *scanState) commitFullBatch(ctx context.Context) error {
|
func (s *scanState) commitFullBatch(ctx context.Context) error {
|
||||||
if len(s.batch) < updateBatchSize {
|
if len(s.batch) < updateBatchSize {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
err := applyBatch(ctx, s.db, s.batch, nil, nil)
|
err := applyBatch(ctx, s.db, s.batch, nil, nil)
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
|
|
||||||
s.batch = s.batch[:0]
|
s.batch = s.batch[:0]
|
||||||
|
|
||||||
return nil
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
// updatePhase writes the scan's tail under one progress display: the
|
// updatePhase writes the scan's tail under one progress display: the
|
||||||
@@ -808,7 +698,7 @@ func unchangedFile(r scanRec) (fileRec, bool, error) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if !fi.Mode().IsRegular() || fi.Size() != r.size ||
|
if !fi.Mode().IsRegular() || fi.Size() != r.size ||
|
||||||
mtimeAfter(fi.ModTime(), r.mtime) {
|
fi.ModTime().Unix() > r.mtime {
|
||||||
return fileRec{}, false, nil
|
return fileRec{}, false, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -819,16 +709,6 @@ func unchangedFile(r scanRec) (fileRec, bool, error) {
|
|||||||
}, true, nil
|
}, true, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// mtimeAfter reports whether mtime a is later than mtime b.
|
|
||||||
// Not a.After(b): time.Time wraps an mtime past year 292 billion; Unix() undoes it.
|
|
||||||
func mtimeAfter(a, b time.Time) bool {
|
|
||||||
if a.Unix() != b.Unix() {
|
|
||||||
return a.Unix() > b.Unix()
|
|
||||||
}
|
|
||||||
|
|
||||||
return a.Nanosecond() > b.Nanosecond()
|
|
||||||
}
|
|
||||||
|
|
||||||
// underAnyRoot reports whether path is any of the roots or lies under
|
// underAnyRoot reports whether path is any of the roots or lies under
|
||||||
// one of them.
|
// one of them.
|
||||||
func underAnyRoot(path string, roots []string) bool {
|
func underAnyRoot(path string, roots []string) bool {
|
||||||
@@ -909,43 +789,11 @@ func sendEvent(ctx context.Context, events chan<- walkEvent,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// operandWarning returns the one-line warning for an operand the walk
|
|
||||||
// does not start from, naming the path and what it is, or "" for one it
|
|
||||||
// does: a regular file, or a directory not named .zfs. Symlinks are
|
|
||||||
// never followed, including as operands.
|
|
||||||
func operandWarning(root string, fi fs.FileInfo) string {
|
|
||||||
var kind string
|
|
||||||
|
|
||||||
switch mode := fi.Mode(); {
|
|
||||||
case mode.IsRegular():
|
|
||||||
return ""
|
|
||||||
case mode.IsDir():
|
|
||||||
if filepath.Base(root) != ".zfs" {
|
|
||||||
return ""
|
|
||||||
}
|
|
||||||
|
|
||||||
kind = ".zfs directory"
|
|
||||||
case mode&fs.ModeSymlink != 0:
|
|
||||||
kind = "symlink"
|
|
||||||
case mode&fs.ModeSocket != 0:
|
|
||||||
kind = "socket"
|
|
||||||
case mode&fs.ModeNamedPipe != 0:
|
|
||||||
kind = "FIFO"
|
|
||||||
case mode&fs.ModeDevice != 0:
|
|
||||||
kind = "device node"
|
|
||||||
default:
|
|
||||||
kind = "non-regular file"
|
|
||||||
}
|
|
||||||
|
|
||||||
return fmt.Sprintf("walk %s: skipping %s operand", root, kind)
|
|
||||||
}
|
|
||||||
|
|
||||||
// seedRoot turns one PATH operand into the walk's starting state: a
|
// seedRoot turns one PATH operand into the walk's starting state: a
|
||||||
// regular-file operand is statted and emitted directly, and a directory
|
// regular-file operand is statted and emitted directly, a directory
|
||||||
// operand becomes an initial job. walkableRoots has already dropped
|
// operand becomes an initial job, and a symlink or other non-regular
|
||||||
// every other operand. One that has changed into something else since
|
// operand yields nothing (symlinks are never followed, including as
|
||||||
// is warned about and skipped here; it is still a root, so the records
|
// operands).
|
||||||
// stored beneath it are deleted as unverified.
|
|
||||||
func seedRoot(ctx context.Context, root string,
|
func seedRoot(ctx context.Context, root string,
|
||||||
events chan<- walkEvent,
|
events chan<- walkEvent,
|
||||||
) []dirJob {
|
) []dirJob {
|
||||||
@@ -959,30 +807,30 @@ func seedRoot(ctx context.Context, root string,
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
warn := operandWarning(root, fi)
|
switch {
|
||||||
if warn != "" {
|
case fi.IsDir():
|
||||||
sendEvent(ctx, events, walkEvent{warn: warn, fail: true})
|
if filepath.Base(root) == ".zfs" {
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
if fi.IsDir() {
|
|
||||||
dev, ok := deviceOfInfo(fi)
|
dev, ok := deviceOfInfo(fi)
|
||||||
|
|
||||||
return []dirJob{{path: root, rootDev: dev, rootDevOK: ok}}
|
return []dirJob{{path: root, rootDev: dev, rootDevOK: ok}}
|
||||||
}
|
case fi.Mode().IsRegular():
|
||||||
|
|
||||||
dev, ino := inodeOfInfo(fi)
|
dev, ino := inodeOfInfo(fi)
|
||||||
|
|
||||||
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
||||||
path: root,
|
path: root,
|
||||||
size: fi.Size(),
|
size: fi.Size(),
|
||||||
mtime: fi.ModTime(),
|
mtime: fi.ModTime().Unix(),
|
||||||
dev: dev,
|
dev: dev,
|
||||||
ino: ino,
|
ino: ino,
|
||||||
}})
|
}})
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
|
default:
|
||||||
|
return nil
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// startWalkWorkers starts the walk worker pool. Each worker processes
|
// startWalkWorkers starts the walk worker pool. Each worker processes
|
||||||
@@ -1081,12 +929,6 @@ func walkOneDir(ctx context.Context, job dirJob, oneFS bool,
|
|||||||
var subs []dirJob
|
var subs []dirJob
|
||||||
|
|
||||||
for _, e := range entries {
|
for _, e := range entries {
|
||||||
// A cancelled scan wants nothing more from this directory: stop
|
|
||||||
// rather than lstat the rest of a large one.
|
|
||||||
if ctx.Err() != nil {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
p := filepath.Join(job.path, e.Name())
|
p := filepath.Join(job.path, e.Name())
|
||||||
|
|
||||||
if e.IsDir() {
|
if e.IsDir() {
|
||||||
@@ -1135,7 +977,7 @@ func emitFile(ctx context.Context, p string, e fs.DirEntry,
|
|||||||
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
||||||
path: p,
|
path: p,
|
||||||
size: info.Size(),
|
size: info.Size(),
|
||||||
mtime: info.ModTime(),
|
mtime: info.ModTime().Unix(),
|
||||||
dev: dev,
|
dev: dev,
|
||||||
ino: ino,
|
ino: ino,
|
||||||
}})
|
}})
|
||||||
|
|||||||
+50
-438
@@ -6,22 +6,15 @@ import (
|
|||||||
"crypto/sha256"
|
"crypto/sha256"
|
||||||
"database/sql"
|
"database/sql"
|
||||||
"encoding/hex"
|
"encoding/hex"
|
||||||
"errors"
|
|
||||||
"fmt"
|
"fmt"
|
||||||
"io"
|
|
||||||
"io/fs"
|
|
||||||
"os"
|
"os"
|
||||||
"os/signal"
|
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"runtime"
|
"runtime"
|
||||||
"slices"
|
"slices"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
"syscall"
|
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"golang.org/x/sys/unix"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// writeFile creates a file with the given content and returns its path.
|
// writeFile creates a file with the given content and returns its path.
|
||||||
@@ -406,7 +399,7 @@ func TestScanContentGate(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := dupeGroups(t, db)
|
groups := collectDupeGroups(recs)
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, same) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, same) {
|
||||||
t.Fatalf("groups = %+v, want only the identical pair %q",
|
t.Fatalf("groups = %+v, want only the identical pair %q",
|
||||||
groups, same)
|
groups, same)
|
||||||
@@ -441,7 +434,7 @@ func TestScanContentAcrossOperands(t *testing.T) {
|
|||||||
want := []string{a, b}
|
want := []string{a, b}
|
||||||
slices.Sort(want)
|
slices.Sort(want)
|
||||||
|
|
||||||
groups := dupeGroups(t, db)
|
groups := collectDupeGroups(recs)
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
||||||
}
|
}
|
||||||
@@ -462,13 +455,13 @@ func TestScanContentWithinOperand(t *testing.T) {
|
|||||||
added := sparseFile(t, dir, "d2", headTailMin)
|
added := sparseFile(t, dir, "d2", headTailMin)
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 3, added: 1, unchanged: 2}) {
|
if st != (scanStats{added: 1, unchanged: 2}) {
|
||||||
t.Fatalf("rescan stats = %+v, want 1 added 2 unchanged", st)
|
t.Fatalf("rescan stats = %+v, want 1 added 2 unchanged", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
want := []string{stored, added}
|
want := []string{stored, added}
|
||||||
|
|
||||||
groups := dupeGroups(t, db)
|
groups := collectDupeGroups(dbRecords(t, db))
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
||||||
}
|
}
|
||||||
@@ -508,7 +501,7 @@ func TestScanContentStalePartners(t *testing.T) {
|
|||||||
sparseFile(t, dirB, "changed-copy", headTailMin+1)
|
sparseFile(t, dirB, "changed-copy", headTailMin+1)
|
||||||
|
|
||||||
st := syncTree(t, db, dirB)
|
st := syncTree(t, db, dirB)
|
||||||
if st != (scanStats{walked: 2, added: 2}) {
|
if st != (scanStats{added: 2}) {
|
||||||
t.Errorf("stats = %+v, want 2 added and nothing skipped", st)
|
t.Errorf("stats = %+v, want 2 added and nothing skipped", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -526,7 +519,7 @@ func TestScanContentStalePartners(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if groups := dupeGroups(t, db); len(groups) != 0 {
|
if groups := collectDupeGroups(recs); len(groups) != 0 {
|
||||||
t.Errorf("groups = %+v, want none", groups)
|
t.Errorf("groups = %+v, want none", groups)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -565,7 +558,7 @@ func TestScanContentHashedStalePartners(t *testing.T) {
|
|||||||
b := sparseFile(t, t.TempDir(), "copy", headTailMin)
|
b := sparseFile(t, t.TempDir(), "copy", headTailMin)
|
||||||
|
|
||||||
st := syncTree(t, db, filepath.Dir(b))
|
st := syncTree(t, db, filepath.Dir(b))
|
||||||
if st != (scanStats{walked: 1, added: 1}) {
|
if st != (scanStats{added: 1}) {
|
||||||
t.Errorf("stats = %+v, want 1 added and nothing skipped", st)
|
t.Errorf("stats = %+v, want 1 added and nothing skipped", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -577,71 +570,12 @@ func TestScanContentHashedStalePartners(t *testing.T) {
|
|||||||
|
|
||||||
// The stored records lie outside the operand and are left as they
|
// The stored records lie outside the operand and are left as they
|
||||||
// are, so they still group with each other, but not with the copy.
|
// are, so they still group with each other, but not with the copy.
|
||||||
groups := dupeGroups(t, db)
|
groups := collectDupeGroups(recs)
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) {
|
||||||
t.Errorf("groups = %+v, want only the stored pair %q", groups, stored)
|
t.Errorf("groups = %+v, want only the stored pair %q", groups, stored)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestScanContentSameSecondRewrite is TestScanContentStalePartners for
|
|
||||||
// a stored file rewritten in place at the same size with an mtime later
|
|
||||||
// in the same second than recorded: the file counts as changed, so
|
|
||||||
// neither it nor its match inside the operand is read.
|
|
||||||
func TestScanContentSameSecondRewrite(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
db := openTestDB(t)
|
|
||||||
dirA := t.TempDir()
|
|
||||||
changed := sparseFileWithoutMatch(t, dirA, "changed", headTailMin)
|
|
||||||
|
|
||||||
first := time.Date(2026, 1, 2, 3, 4, 5, 100_000_000, time.UTC)
|
|
||||||
|
|
||||||
err := os.Chtimes(changed, first, first)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
syncTree(t, db, dirA)
|
|
||||||
|
|
||||||
before := dbRecords(t, db)
|
|
||||||
|
|
||||||
// Rewrite one byte in place, keeping the size.
|
|
||||||
pokeAt(t, changed, headTailMin/2, []byte{1})
|
|
||||||
|
|
||||||
later := first.Add(500 * time.Millisecond)
|
|
||||||
|
|
||||||
err = os.Chtimes(changed, later, later)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
dirB := t.TempDir()
|
|
||||||
sparseFile(t, dirB, "changed-copy", headTailMin)
|
|
||||||
|
|
||||||
st := syncTree(t, db, dirB)
|
|
||||||
if st != (scanStats{walked: 1, added: 1}) {
|
|
||||||
t.Errorf("stats = %+v, want 1 added and nothing skipped", st)
|
|
||||||
}
|
|
||||||
|
|
||||||
recs := dbRecords(t, db)
|
|
||||||
for _, r := range recs {
|
|
||||||
if r.content != "" {
|
|
||||||
t.Errorf("%s: content = %q, want none: its only match is stale",
|
|
||||||
r.path, r.content)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
for _, old := range before {
|
|
||||||
if r := recordByPath(t, recs, old.path); r != old {
|
|
||||||
t.Errorf("record = %+v, want it left as %+v", r, old)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if groups := dupeGroups(t, db); len(groups) != 0 {
|
|
||||||
t.Errorf("groups = %+v, want none", groups)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestScanContentReadFailure checks that a failed content read is
|
// TestScanContentReadFailure checks that a failed content read is
|
||||||
// counted as skipped and leaves the record without a content hash, and
|
// counted as skipped and leaves the record without a content hash, and
|
||||||
// that a later scan tries the read again.
|
// that a later scan tries the read again.
|
||||||
@@ -664,7 +598,7 @@ func TestScanContentReadFailure(t *testing.T) {
|
|||||||
b := sparseFile(t, dirB, "b", headTailMin)
|
b := sparseFile(t, dirB, "b", headTailMin)
|
||||||
|
|
||||||
st := syncTree(t, db, dirB)
|
st := syncTree(t, db, dirB)
|
||||||
if st != (scanStats{walked: 1, added: 1, skipped: 1}) {
|
if st != (scanStats{added: 1, skipped: 1}) {
|
||||||
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -678,14 +612,14 @@ func TestScanContentReadFailure(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
st = syncTree(t, db, dirB)
|
st = syncTree(t, db, dirB)
|
||||||
if st != (scanStats{walked: 1, unchanged: 1}) {
|
if st != (scanStats{unchanged: 1}) {
|
||||||
t.Fatalf("rescan stats = %+v, want 1 unchanged", st)
|
t.Fatalf("rescan stats = %+v, want 1 unchanged", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
want := []string{a, b}
|
want := []string{a, b}
|
||||||
slices.Sort(want)
|
slices.Sort(want)
|
||||||
|
|
||||||
groups := dupeGroups(t, db)
|
groups := collectDupeGroups(dbRecords(t, db))
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
t.Fatalf("groups = %+v, want the pair %q after the retry",
|
t.Fatalf("groups = %+v, want the pair %q after the retry",
|
||||||
groups, want)
|
groups, want)
|
||||||
@@ -724,7 +658,7 @@ func TestScanContentCheckError(t *testing.T) {
|
|||||||
b := sparseFile(t, t.TempDir(), "b", headTailMin)
|
b := sparseFile(t, t.TempDir(), "b", headTailMin)
|
||||||
|
|
||||||
st := syncTree(t, db, filepath.Dir(b))
|
st := syncTree(t, db, filepath.Dir(b))
|
||||||
if st != (scanStats{walked: 1, added: 1, skipped: 1}) {
|
if st != (scanStats{added: 1, skipped: 1}) {
|
||||||
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -752,7 +686,7 @@ func TestScanContentHardlinks(t *testing.T) {
|
|||||||
c := sparseFile(t, dir, "copy", headTailMin)
|
c := sparseFile(t, dir, "copy", headTailMin)
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 3, added: 3}) {
|
if st != (scanStats{added: 3}) {
|
||||||
t.Fatalf("stats = %+v, want 3 added", st)
|
t.Fatalf("stats = %+v, want 3 added", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -844,8 +778,8 @@ func TestWalk(t *testing.T) {
|
|||||||
t.Errorf("%s: size = %d, want 1..3", r.path, r.size)
|
t.Errorf("%s: size = %d, want 1..3", r.path, r.size)
|
||||||
}
|
}
|
||||||
|
|
||||||
if r.mtime.Unix() <= 0 {
|
if r.mtime <= 0 {
|
||||||
t.Errorf("%s: mtime = %v, want after 1970", r.path, r.mtime)
|
t.Errorf("%s: mtime = %d, want positive", r.path, r.mtime)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -922,11 +856,9 @@ func TestWalkFileAndSymlinkOperands(t *testing.T) {
|
|||||||
t.Fatalf("file operand: recs = %+v, errs = %d", recs, errs)
|
t.Fatalf("file operand: recs = %+v, errs = %d", recs, errs)
|
||||||
}
|
}
|
||||||
|
|
||||||
// A symlink operand that reaches the walk (it became one after
|
// A symlink operand is not followed and yields nothing.
|
||||||
// walkableRoots checked it) is not followed: it yields a warning and
|
|
||||||
// no records.
|
|
||||||
recs, errs = collectWalk(t, []string{link}, false, 2)
|
recs, errs = collectWalk(t, []string{link}, false, 2)
|
||||||
if errs != 1 || len(recs) != 0 {
|
if errs != 0 || len(recs) != 0 {
|
||||||
t.Fatalf("symlink operand: recs = %+v, errs = %d", recs, errs)
|
t.Fatalf("symlink operand: recs = %+v, errs = %d", recs, errs)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -975,159 +907,6 @@ func TestDeviceOfInfo(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// dirEntryFor returns the fs.DirEntry for name within dir, obtained via
|
|
||||||
// the same os.ReadDir the walk uses, so it carries a real Info().
|
|
||||||
func dirEntryFor(t *testing.T, dir, name string) fs.DirEntry {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
entries, err := os.ReadDir(dir)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
for _, e := range entries {
|
|
||||||
if e.Name() == name {
|
|
||||||
return e
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
t.Fatalf("entry %q not found in %q", name, dir)
|
|
||||||
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// callSubdirJob runs subdirJob against e under parent, collecting any
|
|
||||||
// warning events it emits (subdirJob emits at most one).
|
|
||||||
func callSubdirJob(t *testing.T, p string, e fs.DirEntry,
|
|
||||||
parent dirJob, oneFS bool,
|
|
||||||
) (dirJob, bool, []walkEvent) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
events := make(chan walkEvent, 1)
|
|
||||||
job, ok := subdirJob(t.Context(), p, e, parent, oneFS, events)
|
|
||||||
close(events)
|
|
||||||
|
|
||||||
var evs []walkEvent
|
|
||||||
for ev := range events {
|
|
||||||
evs = append(evs, ev)
|
|
||||||
}
|
|
||||||
|
|
||||||
return job, ok, evs
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestSubdirJobOneFilesystem exercises the -x boundary check in
|
|
||||||
// subdirJob directly, so no second real filesystem is needed. The
|
|
||||||
// subdirectory's real device is compared against a fabricated operand
|
|
||||||
// device.
|
|
||||||
func TestSubdirJobOneFilesystem(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
dir := t.TempDir()
|
|
||||||
|
|
||||||
sub := filepath.Join(dir, "sub")
|
|
||||||
|
|
||||||
err := os.Mkdir(sub, 0o750)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
info, err := os.Lstat(sub)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
dev, ok := deviceOfInfo(info)
|
|
||||||
if !ok {
|
|
||||||
t.Skip("platform exposes no device id")
|
|
||||||
}
|
|
||||||
|
|
||||||
// A device the subdirectory is not on, standing in for an operand
|
|
||||||
// rooted on a different filesystem.
|
|
||||||
otherDev := dev + 1
|
|
||||||
e := dirEntryFor(t, dir, "sub")
|
|
||||||
|
|
||||||
cases := []struct {
|
|
||||||
name string
|
|
||||||
oneFS bool
|
|
||||||
parent dirJob
|
|
||||||
wantOK bool
|
|
||||||
}{
|
|
||||||
// -x on, subdirectory on a different device than its operand:
|
|
||||||
// descent is refused.
|
|
||||||
{"reject across boundary", true,
|
|
||||||
dirJob{rootDev: otherDev, rootDevOK: true}, false},
|
|
||||||
// -x on but the operand's own device is unknown: the boundary
|
|
||||||
// check is bypassed and descent proceeds.
|
|
||||||
{"bypass when root device unknown", true,
|
|
||||||
dirJob{rootDev: otherDev, rootDevOK: false}, true},
|
|
||||||
// Default (no -x): boundaries are crossed even onto a different
|
|
||||||
// device.
|
|
||||||
{"cross by default", false,
|
|
||||||
dirJob{rootDev: otherDev, rootDevOK: true}, true},
|
|
||||||
}
|
|
||||||
|
|
||||||
for _, tc := range cases {
|
|
||||||
t.Run(tc.name, func(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
job, ok, evs := callSubdirJob(t, sub, e, tc.parent, tc.oneFS)
|
|
||||||
if ok != tc.wantOK {
|
|
||||||
t.Fatalf("accepted = %v, want %v", ok, tc.wantOK)
|
|
||||||
}
|
|
||||||
|
|
||||||
if len(evs) != 0 {
|
|
||||||
t.Fatalf("unexpected events: %+v", evs)
|
|
||||||
}
|
|
||||||
|
|
||||||
// The accepted job must carry the operand's device down, or -x
|
|
||||||
// stops checking below the first level.
|
|
||||||
want := dirJob{
|
|
||||||
path: sub,
|
|
||||||
rootDev: tc.parent.rootDev,
|
|
||||||
rootDevOK: tc.parent.rootDevOK,
|
|
||||||
}
|
|
||||||
if ok && job != want {
|
|
||||||
t.Fatalf("job = %+v, want %+v", job, want)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// errInfoUnavailable is returned by errDirEntry.Info().
|
|
||||||
var errInfoUnavailable = errors.New("info unavailable")
|
|
||||||
|
|
||||||
// errDirEntry is a directory entry whose Info() always fails, driving
|
|
||||||
// subdirJob's stat-error branch deterministically.
|
|
||||||
type errDirEntry struct{ name string }
|
|
||||||
|
|
||||||
func (e errDirEntry) Name() string { return e.name }
|
|
||||||
func (errDirEntry) IsDir() bool { return true }
|
|
||||||
func (errDirEntry) Type() fs.FileMode {
|
|
||||||
return fs.ModeDir
|
|
||||||
}
|
|
||||||
|
|
||||||
func (errDirEntry) Info() (fs.FileInfo, error) {
|
|
||||||
return nil, errInfoUnavailable
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestSubdirJobStatError asserts that when a subdirectory's Info()
|
|
||||||
// fails under -x, subdirJob warns and refuses descent.
|
|
||||||
func TestSubdirJobStatError(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
p := "/does/not/matter/sub"
|
|
||||||
|
|
||||||
_, ok, evs := callSubdirJob(t, p, errDirEntry{name: "sub"},
|
|
||||||
dirJob{rootDev: 1, rootDevOK: true}, true)
|
|
||||||
if ok {
|
|
||||||
t.Fatal("descent accepted after stat error, want refused")
|
|
||||||
}
|
|
||||||
|
|
||||||
if len(evs) != 1 || !evs[0].fail || !strings.Contains(evs[0].warn, p) {
|
|
||||||
t.Fatalf("want one warning naming the path, got %+v", evs)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// buildSmokeTree recreates the README smoke-test filesystem layout
|
// buildSmokeTree recreates the README smoke-test filesystem layout
|
||||||
// with deterministic content and returns the tree root.
|
// with deterministic content and returns the tree root.
|
||||||
func buildSmokeTree(t *testing.T) string {
|
func buildSmokeTree(t *testing.T) string {
|
||||||
@@ -1174,16 +953,11 @@ func syncTree(t *testing.T, db *sql.DB, roots ...string) scanStats {
|
|||||||
return st
|
return st
|
||||||
}
|
}
|
||||||
|
|
||||||
// dbRecords returns every record currently in the database, in path
|
// dbRecords returns every record currently in the database.
|
||||||
// order.
|
|
||||||
func dbRecords(t *testing.T, db *sql.DB) []scanRec {
|
func dbRecords(t *testing.T, db *sql.DB) []scanRec {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
var recs []scanRec
|
recs, err := loadFileRows(t.Context(), db)
|
||||||
|
|
||||||
err := loadFileRows(t.Context(), db, func(r scanRec) {
|
|
||||||
recs = append(recs, r)
|
|
||||||
})
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
@@ -1220,10 +994,10 @@ func recordPaths(recs []scanRec) []string {
|
|||||||
|
|
||||||
// assertSmokeDupeGroups checks the file-level duplicate groups for the
|
// assertSmokeDupeGroups checks the file-level duplicate groups for the
|
||||||
// smoke tree rooted at dir.
|
// smoke tree rooted at dir.
|
||||||
func assertSmokeDupeGroups(t *testing.T, dir string, db *sql.DB) {
|
func assertSmokeDupeGroups(t *testing.T, dir string, parsed []scanRec) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
groups := dupeGroups(t, db)
|
groups := collectDupeGroups(parsed)
|
||||||
if len(groups) != 5 {
|
if len(groups) != 5 {
|
||||||
t.Fatalf("len(groups) = %d, want 5", len(groups))
|
t.Fatalf("len(groups) = %d, want 5", len(groups))
|
||||||
}
|
}
|
||||||
@@ -1248,10 +1022,11 @@ func assertSmokeDupeGroups(t *testing.T, dir string, db *sql.DB) {
|
|||||||
|
|
||||||
// assertSmokeTreeGroups checks the duplicate-tree groups for the smoke
|
// assertSmokeTreeGroups checks the duplicate-tree groups for the smoke
|
||||||
// tree rooted at dir.
|
// tree rooted at dir.
|
||||||
func assertSmokeTreeGroups(t *testing.T, dir string, db *sql.DB) {
|
func assertSmokeTreeGroups(t *testing.T, dir string, parsed []scanRec) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
super, dirs := dbTree(t, db)
|
super, dirs := buildHierarchy(parsed)
|
||||||
|
super.compute()
|
||||||
|
|
||||||
tg := collectTreeGroups(dirs, super)
|
tg := collectTreeGroups(dirs, super)
|
||||||
if len(tg) != 1 {
|
if len(tg) != 1 {
|
||||||
@@ -1276,7 +1051,7 @@ func TestScanPipeline(t *testing.T) {
|
|||||||
db := openTestDB(t)
|
db := openTestDB(t)
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: smokeTreeFiles, added: smokeTreeFiles}) {
|
if st != (scanStats{added: smokeTreeFiles}) {
|
||||||
t.Fatalf("stats = %+v, want %d added only", st, smokeTreeFiles)
|
t.Fatalf("stats = %+v, want %d added only", st, smokeTreeFiles)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1285,8 +1060,8 @@ func TestScanPipeline(t *testing.T) {
|
|||||||
t.Fatalf("len(records) = %d, want %d", len(parsed), smokeTreeFiles)
|
t.Fatalf("len(records) = %d, want %d", len(parsed), smokeTreeFiles)
|
||||||
}
|
}
|
||||||
|
|
||||||
assertSmokeDupeGroups(t, dir, db)
|
assertSmokeDupeGroups(t, dir, parsed)
|
||||||
assertSmokeTreeGroups(t, dir, db)
|
assertSmokeTreeGroups(t, dir, parsed)
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestSyncScanUnchangedReuse(t *testing.T) {
|
func TestSyncScanUnchangedReuse(t *testing.T) {
|
||||||
@@ -1299,7 +1074,7 @@ func TestSyncScanUnchangedReuse(t *testing.T) {
|
|||||||
writeFile(t, dir, "b.bin", pattern(2, 600))
|
writeFile(t, dir, "b.bin", pattern(2, 600))
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 2, added: 2}) {
|
if st != (scanStats{added: 2}) {
|
||||||
t.Fatalf("first scan stats = %+v, want 2 added", st)
|
t.Fatalf("first scan stats = %+v, want 2 added", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1313,7 +1088,7 @@ func TestSyncScanUnchangedReuse(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
st = syncTree(t, db, dir)
|
st = syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 2, unchanged: 2}) {
|
if st != (scanStats{unchanged: 2}) {
|
||||||
t.Fatalf("rescan stats = %+v, want 2 unchanged", st)
|
t.Fatalf("rescan stats = %+v, want 2 unchanged", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1342,176 +1117,15 @@ func TestSyncScanMtimeBump(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 1, updated: 1}) {
|
if st != (scanStats{updated: 1}) {
|
||||||
t.Fatalf("mtime-bump stats = %+v, want 1 updated", st)
|
t.Fatalf("mtime-bump stats = %+v, want 1 updated", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
if r := recordByPath(t, dbRecords(t, db), a); !r.mtime.Equal(future) {
|
if r := recordByPath(t, dbRecords(t, db), a); r.mtime != future.Unix() {
|
||||||
t.Fatalf("mtime = %v, want %v", r.mtime, future)
|
t.Fatalf("mtime = %d, want %d", r.mtime, future.Unix())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// assertWholeFileHashed fails unless the record for path holds the
|
|
||||||
// whole-file hash of data as its head, tail, and content.
|
|
||||||
func assertWholeFileHashed(t *testing.T, db *sql.DB, path string,
|
|
||||||
data []byte,
|
|
||||||
) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
r := recordByPath(t, dbRecords(t, db), path)
|
|
||||||
if want := hexSum(data); r.head != want || r.tail != want ||
|
|
||||||
r.content != want {
|
|
||||||
t.Fatalf("head, tail, content = %q, %q, %q, want %q for each",
|
|
||||||
r.head, r.tail, r.content, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestSyncScanSameSecondRewrite rewrites a file in place at the same
|
|
||||||
// size with an mtime later in the same second as the recorded one: the
|
|
||||||
// next scan must notice the change and re-hash the file.
|
|
||||||
func TestSyncScanSameSecondRewrite(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
dir := t.TempDir()
|
|
||||||
db := openTestDB(t)
|
|
||||||
a := writeFile(t, dir, "a.bin", pattern(1, 500))
|
|
||||||
|
|
||||||
// b.bin shares the size of a.bin, so a.bin is hashed.
|
|
||||||
writeFile(t, dir, "b.bin", pattern(2, 500))
|
|
||||||
|
|
||||||
first := time.Date(2026, 1, 2, 3, 4, 5, 100_000_000, time.UTC)
|
|
||||||
|
|
||||||
err := os.Chtimes(a, first, first)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
syncTree(t, db, dir)
|
|
||||||
|
|
||||||
rewritten := pattern(3, 500)
|
|
||||||
writeFile(t, dir, "a.bin", rewritten)
|
|
||||||
|
|
||||||
later := first.Add(500 * time.Millisecond)
|
|
||||||
|
|
||||||
err = os.Chtimes(a, later, later)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
|
||||||
if st != (scanStats{walked: 2, updated: 1, unchanged: 1}) {
|
|
||||||
t.Fatalf("rescan stats = %+v, want 1 updated 1 unchanged", st)
|
|
||||||
}
|
|
||||||
|
|
||||||
assertWholeFileHashed(t, db, a, rewritten)
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestSyncScanOperandSameSecondRewrite is TestSyncScanSameSecondRewrite
|
|
||||||
// for files given to scan as operands, which scan stats without reading
|
|
||||||
// their directory.
|
|
||||||
func TestSyncScanOperandSameSecondRewrite(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
dir := t.TempDir()
|
|
||||||
db := openTestDB(t)
|
|
||||||
a := writeFile(t, dir, "a.bin", pattern(1, 500))
|
|
||||||
|
|
||||||
// b.bin shares the size of a.bin, so a.bin is hashed.
|
|
||||||
b := writeFile(t, dir, "b.bin", pattern(2, 500))
|
|
||||||
|
|
||||||
first := time.Date(2026, 1, 2, 3, 4, 5, 100_000_000, time.UTC)
|
|
||||||
|
|
||||||
err := os.Chtimes(a, first, first)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
syncTree(t, db, a, b)
|
|
||||||
|
|
||||||
rewritten := pattern(3, 500)
|
|
||||||
writeFile(t, dir, "a.bin", rewritten)
|
|
||||||
|
|
||||||
later := first.Add(500 * time.Millisecond)
|
|
||||||
|
|
||||||
err = os.Chtimes(a, later, later)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
st := syncTree(t, db, a, b)
|
|
||||||
if st != (scanStats{walked: 2, updated: 1, unchanged: 1}) {
|
|
||||||
t.Fatalf("rescan stats = %+v, want 1 updated 1 unchanged", st)
|
|
||||||
}
|
|
||||||
|
|
||||||
assertWholeFileHashed(t, db, a, rewritten)
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestSyncScanRewriteAfter2262 runs assertLateRewriteRehashed with an
|
|
||||||
// mtime after 2262, a time too late to count in nanoseconds in an int64.
|
|
||||||
func TestSyncScanRewriteAfter2262(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
assertLateRewriteRehashed(t, time.Date(2300, 1, 2, 3, 4, 5, 0, time.UTC))
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestSyncScanRewritePastTimeLimit runs assertLateRewriteRehashed with an
|
|
||||||
// mtime one second past the latest a time.Time holds without wrapping it
|
|
||||||
// to a time far in the past.
|
|
||||||
func TestSyncScanRewritePastTimeLimit(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
assertLateRewriteRehashed(t, time.Unix(9223371974719179008, 0))
|
|
||||||
}
|
|
||||||
|
|
||||||
// assertLateRewriteRehashed scans a directory, rewrites a file in it in
|
|
||||||
// place at the same size, sets its mtime to late, and fails unless the
|
|
||||||
// next scan re-hashes the file. It skips where late does not fit the
|
|
||||||
// platform's timespec or the filesystem does not store it.
|
|
||||||
func assertLateRewriteRehashed(t *testing.T, late time.Time) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
dir := t.TempDir()
|
|
||||||
db := openTestDB(t)
|
|
||||||
a := writeFile(t, dir, "a.bin", pattern(1, 500))
|
|
||||||
|
|
||||||
// b.bin shares the size of a.bin, so a.bin is hashed.
|
|
||||||
writeFile(t, dir, "b.bin", pattern(2, 500))
|
|
||||||
|
|
||||||
syncTree(t, db, dir)
|
|
||||||
|
|
||||||
rewritten := pattern(3, 500)
|
|
||||||
writeFile(t, dir, "a.bin", rewritten)
|
|
||||||
|
|
||||||
// os.Chtimes cannot set such a time: it converts through UnixNano.
|
|
||||||
ts, err := unix.TimeToTimespec(late)
|
|
||||||
if err != nil {
|
|
||||||
t.Skipf("an mtime %d seconds after 1970 does not fit this platform's "+
|
|
||||||
"timespec: %v", late.Unix(), err)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = unix.UtimesNano(a, []unix.Timespec{ts, ts})
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
fi, err := os.Lstat(a)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
if fi.ModTime().Unix() != late.Unix() {
|
|
||||||
t.Skipf("the filesystem stored the mtime as %d seconds after 1970, "+
|
|
||||||
"not %d", fi.ModTime().Unix(), late.Unix())
|
|
||||||
}
|
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
|
||||||
if st != (scanStats{walked: 2, updated: 1, unchanged: 1}) {
|
|
||||||
t.Fatalf("rescan stats = %+v, want 1 updated 1 unchanged", st)
|
|
||||||
}
|
|
||||||
|
|
||||||
assertWholeFileHashed(t, db, a, rewritten)
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestSyncScanAddRemove(t *testing.T) {
|
func TestSyncScanAddRemove(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
@@ -1531,7 +1145,7 @@ func TestSyncScanAddRemove(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 2, added: 1, removed: 1, unchanged: 1}) {
|
if st != (scanStats{added: 1, removed: 1, unchanged: 1}) {
|
||||||
t.Fatalf("add/remove stats = %+v, want 1 added 1 removed 1 unchanged",
|
t.Fatalf("add/remove stats = %+v, want 1 added 1 removed 1 unchanged",
|
||||||
st)
|
st)
|
||||||
}
|
}
|
||||||
@@ -1557,7 +1171,9 @@ func TestSyncScanSizeChange(t *testing.T) {
|
|||||||
|
|
||||||
writeFile(t, dir, "f", pattern(1, 200))
|
writeFile(t, dir, "f", pattern(1, 200))
|
||||||
|
|
||||||
err := os.Chtimes(p, old.mtime, old.mtime)
|
mt := time.Unix(old.mtime, 0)
|
||||||
|
|
||||||
|
err := os.Chtimes(p, mt, mt)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
@@ -1646,7 +1262,7 @@ func TestSyncScanOverlappingRoots(t *testing.T) {
|
|||||||
// A file reachable via two overlapping operands is deduplicated
|
// A file reachable via two overlapping operands is deduplicated
|
||||||
// by path in the shared walk and processed once.
|
// by path in the shared walk and processed once.
|
||||||
st := syncTree(t, db, dir, filepath.Join(dir, "sub"))
|
st := syncTree(t, db, dir, filepath.Join(dir, "sub"))
|
||||||
if st != (scanStats{walked: 1, added: 1}) {
|
if st != (scanStats{added: 1}) {
|
||||||
t.Fatalf("stats = %+v, want 1 added", st)
|
t.Fatalf("stats = %+v, want 1 added", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1667,7 +1283,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
|||||||
// Neither size is shared, so neither file is read: both records
|
// Neither size is shared, so neither file is read: both records
|
||||||
// are written without hashes and no duplicates are reported.
|
// are written without hashes and no duplicates are reported.
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 2, added: 2}) {
|
if st != (scanStats{added: 2}) {
|
||||||
t.Fatalf("stats = %+v, want 2 added", st)
|
t.Fatalf("stats = %+v, want 2 added", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1679,7 +1295,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if groups := dupeGroups(t, db); len(groups) != 0 {
|
if groups := collectDupeGroups(recs); len(groups) != 0 {
|
||||||
t.Fatalf("groups = %+v, want none from unhashed records", groups)
|
t.Fatalf("groups = %+v, want none from unhashed records", groups)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1689,12 +1305,12 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
|||||||
c := writeFile(t, dir, "c.bin", pattern(1, 500))
|
c := writeFile(t, dir, "c.bin", pattern(1, 500))
|
||||||
|
|
||||||
st = syncTree(t, db, dir)
|
st = syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 3, added: 1, updated: 1, unchanged: 1}) {
|
if st != (scanStats{added: 1, updated: 1, unchanged: 1}) {
|
||||||
t.Fatalf("rescan stats = %+v, want 1 added 1 updated 1 unchanged",
|
t.Fatalf("rescan stats = %+v, want 1 added 1 updated 1 unchanged",
|
||||||
st)
|
st)
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := dupeGroups(t, db)
|
groups := collectDupeGroups(dbRecords(t, db))
|
||||||
if len(groups) != 1 {
|
if len(groups) != 1 {
|
||||||
t.Fatalf("groups = %+v, want the a/c pair", groups)
|
t.Fatalf("groups = %+v, want the a/c pair", groups)
|
||||||
}
|
}
|
||||||
@@ -1730,7 +1346,8 @@ func TestTreesUnhashedNeverEqual(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
for name, unknown := range cases {
|
for name, unknown := range cases {
|
||||||
super, dirs := treeOf(t, append(slices.Clone(shared), unknown...))
|
super, dirs := buildHierarchy(append(slices.Clone(shared), unknown...))
|
||||||
|
super.compute()
|
||||||
|
|
||||||
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
|
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
|
||||||
t.Errorf("%s: tree groups = %d, want 0 (the files may differ)",
|
t.Errorf("%s: tree groups = %d, want 0 (the files may differ)",
|
||||||
@@ -1753,7 +1370,7 @@ func TestScanHardlinksReadOnce(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 2, added: 2}) {
|
if st != (scanStats{added: 2}) {
|
||||||
t.Fatalf("stats = %+v, want 2 added", st)
|
t.Fatalf("stats = %+v, want 2 added", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1767,7 +1384,7 @@ func TestScanHardlinksReadOnce(t *testing.T) {
|
|||||||
t.Fatalf("hardlink hashes differ: %+v vs %+v", ra, rb)
|
t.Fatalf("hardlink hashes differ: %+v vs %+v", ra, rb)
|
||||||
}
|
}
|
||||||
|
|
||||||
if groups := dupeGroups(t, db); len(groups) != 1 {
|
if groups := collectDupeGroups(recs); len(groups) != 1 {
|
||||||
t.Fatalf("groups = %+v, want the hardlink pair", groups)
|
t.Fatalf("groups = %+v, want the hardlink pair", groups)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1875,13 +1492,6 @@ func injectWriteFailure(t *testing.T, path string) {
|
|||||||
func baselineGoroutines(t *testing.T) int {
|
func baselineGoroutines(t *testing.T) int {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
// The first scan command in a process starts os/signal's goroutine,
|
|
||||||
// which never exits. Start it now, so the baseline counts it
|
|
||||||
// instead of the scan seeming to leave it behind.
|
|
||||||
ch := make(chan os.Signal, 1)
|
|
||||||
signal.Notify(ch, syscall.SIGINT)
|
|
||||||
signal.Stop(ch)
|
|
||||||
|
|
||||||
deadline := time.Now().Add(goroutineSettle)
|
deadline := time.Now().Add(goroutineSettle)
|
||||||
last := runtime.NumGoroutine()
|
last := runtime.NumGoroutine()
|
||||||
|
|
||||||
@@ -1938,7 +1548,7 @@ func TestScanHashWriteFailureUnwindsPool(t *testing.T) {
|
|||||||
|
|
||||||
code := run([]string{
|
code := run([]string{
|
||||||
cmdScan, "--workers", strconv.Itoa(hashLeakWorkers), dir,
|
cmdScan, "--workers", strconv.Itoa(hashLeakWorkers), dir,
|
||||||
}, io.Discard, &stderr)
|
}, &stderr)
|
||||||
if code != exitFatal {
|
if code != exitFatal {
|
||||||
t.Fatalf("run(scan) = %d, want %d; stderr: %s",
|
t.Fatalf("run(scan) = %d, want %d; stderr: %s",
|
||||||
code, exitFatal, stderr.String())
|
code, exitFatal, stderr.String())
|
||||||
@@ -2015,8 +1625,10 @@ func TestReportsNeverTouchFilesystem(t *testing.T) {
|
|||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
|
|
||||||
assertSmokeDupeGroups(t, dir, db)
|
recs := dbRecords(t, db)
|
||||||
assertSmokeTreeGroups(t, dir, db)
|
|
||||||
|
assertSmokeDupeGroups(t, dir, recs)
|
||||||
|
assertSmokeTreeGroups(t, dir, recs)
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestUnderRoot(t *testing.T) {
|
func TestUnderRoot(t *testing.T) {
|
||||||
|
|||||||
+23
-82
@@ -2,21 +2,15 @@
|
|||||||
# script/bootstrap: install all dependencies needed to build and develop
|
# script/bootstrap: install all dependencies needed to build and develop
|
||||||
# this repo. Idempotent: every install is guarded by a check so already
|
# this repo. Idempotent: every install is guarded by a check so already
|
||||||
# installed tools are skipped. Base tooling comes from nix, apt, brew,
|
# installed tools are skipped. Base tooling comes from nix, apt, brew,
|
||||||
# or apk (detected in that order); assumes nothing is present. Node is
|
# or apk (detected in that order); assumes nothing is present (not git,
|
||||||
# used directly if installed; otherwise it is installed at a pinned
|
# make, or go). The linter is NOT installed: golangci-lint runs via
|
||||||
# version via nvm (installing nvm itself first, from a hash-verified
|
# docker only (script/lint), pinned by image digest, so the only lint
|
||||||
# release archive, never curl | sh).
|
# prerequisite is a working docker — which is warned about, not
|
||||||
|
# installed, because everything except linting works without it.
|
||||||
set -eu
|
set -eu
|
||||||
|
|
||||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||||
|
|
||||||
# Pinned versions, 2026-07-06
|
|
||||||
NODE_VERSION="22.17.0"
|
|
||||||
NVM_VERSION="0.40.3"
|
|
||||||
# sha256 of https://github.com/nvm-sh/nvm/archive/refs/tags/v0.40.3.tar.gz
|
|
||||||
NVM_SHA256="5f4d6aaa04a177dc93c985e31dbc411ab6b8c6e1e21d8015dbc1372625fcd1d0"
|
|
||||||
YARN_VERSION="1.22.22"
|
|
||||||
|
|
||||||
PKGMGR=""
|
PKGMGR=""
|
||||||
SUDO=""
|
SUDO=""
|
||||||
APT_UPDATED=""
|
APT_UPDATED=""
|
||||||
@@ -49,7 +43,6 @@ pkg_install() {
|
|||||||
case "$PKGMGR" in
|
case "$PKGMGR" in
|
||||||
nix) nix-env -iA "nixpkgs.$1" ;;
|
nix) nix-env -iA "nixpkgs.$1" ;;
|
||||||
apt)
|
apt)
|
||||||
# Package lists may be empty (fresh images); refresh once per run.
|
|
||||||
if [ -z "$APT_UPDATED" ]; then
|
if [ -z "$APT_UPDATED" ]; then
|
||||||
$SUDO env DEBIAN_FRONTEND=noninteractive apt-get update
|
$SUDO env DEBIAN_FRONTEND=noninteractive apt-get update
|
||||||
APT_UPDATED=1
|
APT_UPDATED=1
|
||||||
@@ -65,82 +58,30 @@ missing() {
|
|||||||
! command -v "$1" >/dev/null 2>&1
|
! command -v "$1" >/dev/null 2>&1
|
||||||
}
|
}
|
||||||
|
|
||||||
# verify_sha256 <file> <expected-hash>
|
|
||||||
verify_sha256() {
|
|
||||||
if command -v sha256sum >/dev/null 2>&1; then
|
|
||||||
actual="$(sha256sum "$1" | cut -d' ' -f1)"
|
|
||||||
else
|
|
||||||
actual="$(shasum -a 256 "$1" | cut -d' ' -f1)"
|
|
||||||
fi
|
|
||||||
if [ "$actual" != "$2" ]; then
|
|
||||||
echo "bootstrap: sha256 mismatch for $1" >&2
|
|
||||||
echo " expected: $2" >&2
|
|
||||||
echo " actual: $actual" >&2
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
}
|
|
||||||
|
|
||||||
# nvm is a bash script; run a command in a bash with nvm loaded
|
|
||||||
nvm_sh() {
|
|
||||||
bash -c ". \"\$HOME/.nvm/nvm.sh\" && $*"
|
|
||||||
}
|
|
||||||
|
|
||||||
ensure_nvm() {
|
|
||||||
[ -s "$HOME/.nvm/nvm.sh" ] && return 0
|
|
||||||
# nvm prerequisites; nvm itself requires bash
|
|
||||||
if missing bash; then pkg_install bash bash bash bash; fi
|
|
||||||
if missing curl; then pkg_install curl curl curl curl; fi
|
|
||||||
if missing git; then pkg_install git git git git; fi
|
|
||||||
tmp="$(mktemp -d)"
|
|
||||||
curl -fsSL -o "$tmp/nvm.tar.gz" \
|
|
||||||
"https://github.com/nvm-sh/nvm/archive/refs/tags/v${NVM_VERSION}.tar.gz"
|
|
||||||
verify_sha256 "$tmp/nvm.tar.gz" "$NVM_SHA256"
|
|
||||||
mkdir -p "$HOME/.nvm"
|
|
||||||
tar -xzf "$tmp/nvm.tar.gz" -C "$HOME/.nvm" --strip-components=1
|
|
||||||
rm -rf "$tmp"
|
|
||||||
}
|
|
||||||
|
|
||||||
ensure_node() {
|
|
||||||
if ! missing node; then return 0; fi
|
|
||||||
ensure_nvm
|
|
||||||
nvm_sh "nvm install $NODE_VERSION"
|
|
||||||
}
|
|
||||||
|
|
||||||
ensure_yarn() {
|
|
||||||
if ! missing yarn; then return 0; fi
|
|
||||||
if ! missing corepack; then
|
|
||||||
corepack enable
|
|
||||||
corepack prepare "yarn@$YARN_VERSION" --activate
|
|
||||||
elif [ -s "$HOME/.nvm/nvm.sh" ]; then
|
|
||||||
nvm_sh "nvm use $NODE_VERSION >/dev/null && corepack enable && \
|
|
||||||
corepack prepare yarn@$YARN_VERSION --activate"
|
|
||||||
else
|
|
||||||
npm install -g "yarn@$YARN_VERSION"
|
|
||||||
fi
|
|
||||||
}
|
|
||||||
|
|
||||||
install_js_deps() {
|
|
||||||
if missing yarn && [ -s "$HOME/.nvm/nvm.sh" ]; then
|
|
||||||
nvm_sh "nvm use $NODE_VERSION >/dev/null && cd \"$ROOT\" && \
|
|
||||||
yarn install --frozen-lockfile"
|
|
||||||
else
|
|
||||||
yarn install --frozen-lockfile
|
|
||||||
fi
|
|
||||||
}
|
|
||||||
|
|
||||||
main() {
|
main() {
|
||||||
cd "$ROOT"
|
cd "$ROOT"
|
||||||
|
|
||||||
if missing make; then pkg_install gnumake make make make; fi
|
# System tooling, deliberately unpinned: these come from the host
|
||||||
|
# package manager and whatever version it ships is what the host
|
||||||
|
# gets, so a presence check is the right check. The repo pins no
|
||||||
|
# system toolchain versions — the Go language version is governed by
|
||||||
|
# go.mod, and builds that must be reproducible run in the Docker
|
||||||
|
# image, whose base images are pinned by digest.
|
||||||
if missing git; then pkg_install git git git git; fi
|
if missing git; then pkg_install git git git git; fi
|
||||||
# Go builds the binary and runs gofmt. Presence is the whole check:
|
if missing make; then pkg_install gnumake make make make; fi
|
||||||
# go.mod names the Go version, and the tests and the linter run in
|
|
||||||
# digest-pinned images.
|
|
||||||
if missing go; then pkg_install go golang go go; fi
|
if missing go; then pkg_install go golang go go; fi
|
||||||
|
|
||||||
ensure_node
|
# Linting runs via docker only (script/lint), so docker is a lint
|
||||||
ensure_yarn
|
# prerequisite rather than something bootstrap installs. Warn, do
|
||||||
install_js_deps
|
# not fail: everything except `make lint` — and, through it,
|
||||||
|
# `make check`, `make docker` and the pre-commit hook — works
|
||||||
|
# without it.
|
||||||
|
if missing docker; then
|
||||||
|
echo "bootstrap: WARNING: docker not found; make lint, make check" >&2
|
||||||
|
echo "bootstrap: and make docker require it. Install docker to" >&2
|
||||||
|
echo "bootstrap: run the linter." >&2
|
||||||
|
fi
|
||||||
|
|
||||||
go mod download
|
go mod download
|
||||||
|
|
||||||
echo "bootstrap complete"
|
echo "bootstrap complete"
|
||||||
|
|||||||
+1
-3
@@ -1,8 +1,6 @@
|
|||||||
#!/bin/sh
|
#!/bin/sh
|
||||||
# script/check: run all checks (test, lint, fmt-check). Our own
|
# script/check: run all checks (test, lint, fmt-check). Our own
|
||||||
# extension to scripts-to-rule-them-all. test and lint are Docker
|
# extension to scripts-to-rule-them-all. Must not modify any files.
|
||||||
# phases; fmt-check is native, because a formatter writes the working
|
|
||||||
# tree. Must not modify any files.
|
|
||||||
set -eu
|
set -eu
|
||||||
|
|
||||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||||
|
|||||||
+25
-21
@@ -1,30 +1,34 @@
|
|||||||
#!/bin/sh
|
#!/bin/sh
|
||||||
# script/cibuild: run the CI build. It bootstraps first: a CI runner
|
# script/cibuild: run the CI build. The Gitea workflow runs this on
|
||||||
# checks out and runs this and nothing else, and script/fmt-check runs
|
# push.
|
||||||
# the formatter on the host, which a pristine checkout cannot do.
|
#
|
||||||
# --no-cache for the same reason as script/docker: the gate phases the
|
# The Dockerfile runs the gates individually as build steps, not the
|
||||||
# final stage depends on are RUN steps, and a cached one is a check that
|
# make check aggregate: the lint stage runs make fmt-check,
|
||||||
# did not run.
|
# script/verify-lint-image-pin, golangci-lint config verify and
|
||||||
|
# golangci-lint run; the build stage, dropped to an unprivileged user,
|
||||||
|
# runs make test and make fmt-check. Neither make lint nor make check
|
||||||
|
# appears, because both reach script/lint, which is itself a docker
|
||||||
|
# build, and a docker build cannot run inside one. Lint is not skipped
|
||||||
|
# by that — the linter is invoked directly in the lint stage, and the
|
||||||
|
# build stage's COPY --from=lint makes that stage a prerequisite, so
|
||||||
|
# BuildKit must finish it first. Between the two stages everything
|
||||||
|
# make check would run has run, which is why a successful build here
|
||||||
|
# implies the repo is green.
|
||||||
|
#
|
||||||
|
# That implication holds only because of CHECK_EPOCH. A COPY layer is
|
||||||
|
# invalidated by changed content, and a merge commit's tree is
|
||||||
|
# byte-identical to the branch head it merges, so without a fresh value
|
||||||
|
# here Docker serves the gate layers from cache and the build reports a
|
||||||
|
# green it never earned. Passing the current epoch invalidates the gate
|
||||||
|
# layers on every run while leaving the pinned base images and
|
||||||
|
# go mod download cached; see the Dockerfile for the placement.
|
||||||
set -eu
|
set -eu
|
||||||
|
|
||||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||||
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
|
||||||
|
|
||||||
main() {
|
main() {
|
||||||
cd "$ROOT"
|
cd "$ROOT"
|
||||||
"$SCRIPT_DIR/bootstrap"
|
docker build --build-arg CHECK_EPOCH="$(date +%s)" .
|
||||||
"$SCRIPT_DIR/check"
|
|
||||||
# The version and the tag each get their own line: a failing
|
|
||||||
# command substitution inside an argument does not trip `set -e`,
|
|
||||||
# so the inline form degrades silently to an empty constant. The
|
|
||||||
# VERSION build argument takes precedence over the version a build
|
|
||||||
# stage derives from the .git in the context.
|
|
||||||
version="$(git describe --tags --always --dirty 2>/dev/null || true)"
|
|
||||||
[ -n "$version" ] || version="unknown"
|
|
||||||
tag="$("$SCRIPT_DIR/projectname")"
|
|
||||||
docker build --no-cache \
|
|
||||||
--build-arg VERSION="$version" \
|
|
||||||
-t "$tag" .
|
|
||||||
}
|
}
|
||||||
|
|
||||||
main "$@"
|
main "$@"
|
||||||
|
|||||||
+13
-14
@@ -1,8 +1,14 @@
|
|||||||
#!/bin/sh
|
#!/bin/sh
|
||||||
# script/docker: build the Docker image tagged with the project name.
|
# script/docker: build the Docker image tagged with the project name.
|
||||||
# Identical in all repos; the tag comes from script/projectname.
|
# The tag comes from script/projectname.
|
||||||
# --no-cache because the gate phases the final stage depends on are RUN
|
#
|
||||||
# steps, and a cached one is a check that did not run.
|
# CHECK_EPOCH is passed for the same reason script/cibuild passes it:
|
||||||
|
# without it Docker serves the Dockerfile's gate layers from cache on an
|
||||||
|
# unchanged tree and this exits 0 having run neither the lint stage's
|
||||||
|
# gates nor the builder stage's test and fmt-check gates. This is the
|
||||||
|
# set of gates a developer or reviewer runs by hand, so a cached pass
|
||||||
|
# here is the most misleading result the repo can produce. Dependency
|
||||||
|
# layers sit above the ARG and stay cached.
|
||||||
set -eu
|
set -eu
|
||||||
|
|
||||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||||
@@ -10,17 +16,10 @@ ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
|||||||
|
|
||||||
main() {
|
main() {
|
||||||
cd "$ROOT"
|
cd "$ROOT"
|
||||||
# The version and the tag each get their own line: a failing
|
docker build \
|
||||||
# command substitution inside an argument does not trip `set -e`,
|
--build-arg CHECK_EPOCH="$(date +%s)" \
|
||||||
# so the inline form degrades silently to an empty constant. The
|
-t "$("$SCRIPT_DIR/projectname")" \
|
||||||
# VERSION build argument takes precedence over the version a build
|
.
|
||||||
# stage derives from the .git in the context.
|
|
||||||
version="$(git describe --tags --always --dirty 2>/dev/null || true)"
|
|
||||||
[ -n "$version" ] || version="unknown"
|
|
||||||
tag="$("$SCRIPT_DIR/projectname")"
|
|
||||||
docker build --no-cache \
|
|
||||||
--build-arg VERSION="$version" \
|
|
||||||
-t "$tag" .
|
|
||||||
}
|
}
|
||||||
|
|
||||||
main "$@"
|
main "$@"
|
||||||
|
|||||||
-20
@@ -4,29 +4,9 @@ set -eu
|
|||||||
|
|
||||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||||
|
|
||||||
# Must match the pin in script/bootstrap.
|
|
||||||
NODE_VERSION="22.17.0"
|
|
||||||
|
|
||||||
# script/bootstrap installs node and yarn under nvm and leaves neither
|
|
||||||
# on the PATH of the shell that called it, so resolve the pinned
|
|
||||||
# toolchain here the way bootstrap's own install step does. nvm is a
|
|
||||||
# bash script, hence the subshell.
|
|
||||||
run_yarn() {
|
|
||||||
if command -v yarn >/dev/null 2>&1; then
|
|
||||||
exec yarn "$@"
|
|
||||||
fi
|
|
||||||
if [ ! -s "$HOME/.nvm/nvm.sh" ]; then
|
|
||||||
echo "fmt: no yarn; run script/bootstrap first" >&2
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
exec bash -c '. "$HOME/.nvm/nvm.sh" && nvm use "$1" >/dev/null &&
|
|
||||||
shift && exec yarn "$@"' bash "$NODE_VERSION" "$@"
|
|
||||||
}
|
|
||||||
|
|
||||||
main() {
|
main() {
|
||||||
cd "$ROOT"
|
cd "$ROOT"
|
||||||
gofmt -s -w .
|
gofmt -s -w .
|
||||||
run_yarn run prettier --write '**/*.md' --tab-width 4 --prose-wrap always
|
|
||||||
}
|
}
|
||||||
|
|
||||||
main "$@"
|
main "$@"
|
||||||
|
|||||||
+4
-36
@@ -1,50 +1,18 @@
|
|||||||
#!/bin/sh
|
#!/bin/sh
|
||||||
# script/fmt-check: check formatting (read-only).
|
# script/fmt-check: check formatting (read-only). Same scope as
|
||||||
|
# script/fmt, but fails instead of writing.
|
||||||
set -eu
|
set -eu
|
||||||
|
|
||||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||||
|
|
||||||
# Must match the pin in script/bootstrap.
|
|
||||||
NODE_VERSION="22.17.0"
|
|
||||||
|
|
||||||
# script/bootstrap installs node and yarn under nvm and leaves neither
|
|
||||||
# on the PATH of the shell that called it, so resolve the pinned
|
|
||||||
# toolchain here the way bootstrap's own install step does. nvm is a
|
|
||||||
# bash script, hence the subshell.
|
|
||||||
run_yarn() {
|
|
||||||
if command -v yarn >/dev/null 2>&1; then
|
|
||||||
exec yarn "$@"
|
|
||||||
fi
|
|
||||||
if [ ! -s "$HOME/.nvm/nvm.sh" ]; then
|
|
||||||
echo "fmt-check: no yarn; run script/bootstrap first" >&2
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
exec bash -c '. "$HOME/.nvm/nvm.sh" && nvm use "$1" >/dev/null &&
|
|
||||||
shift && exec yarn "$@"' bash "$NODE_VERSION" "$@"
|
|
||||||
}
|
|
||||||
|
|
||||||
main() {
|
main() {
|
||||||
cd "$ROOT"
|
cd "$ROOT"
|
||||||
status=0
|
files="$(gofmt -s -l .)"
|
||||||
|
|
||||||
# gofmt and prettier both run every time, so the output names each
|
|
||||||
# one that fails. Under set -e a bare assignment would end the
|
|
||||||
# script when gofmt fails (a Go file it cannot parse).
|
|
||||||
if ! files="$(gofmt -s -l .)"; then
|
|
||||||
echo "gofmt: failed; see its errors above" >&2
|
|
||||||
status=1
|
|
||||||
fi
|
|
||||||
if [ -n "$files" ]; then
|
if [ -n "$files" ]; then
|
||||||
echo "gofmt: files not formatted:" >&2
|
echo "gofmt: files not formatted:" >&2
|
||||||
echo "$files" >&2
|
echo "$files" >&2
|
||||||
status=1
|
exit 1
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# run_yarn ends in exec; the subshell returns here afterwards.
|
|
||||||
(run_yarn run prettier --check '**/*.md' --tab-width 4 --prose-wrap always) ||
|
|
||||||
status=1
|
|
||||||
|
|
||||||
exit "$status"
|
|
||||||
}
|
}
|
||||||
|
|
||||||
main "$@"
|
main "$@"
|
||||||
|
|||||||
@@ -1,15 +1,19 @@
|
|||||||
#!/bin/sh
|
#!/bin/sh
|
||||||
# script/install-precommit: install the git pre-commit hook that runs
|
# script/install-precommit: install the git pre-commit hook that runs
|
||||||
# script/precommit. Our own extension to scripts-to-rule-them-all.
|
# script/precommit. Our own extension to scripts-to-rule-them-all.
|
||||||
|
# Hooks are shared between the main checkout and all worktrees, so
|
||||||
|
# resolve the common git dir instead of assuming .git is a directory.
|
||||||
set -eu
|
set -eu
|
||||||
|
|
||||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||||
|
|
||||||
main() {
|
main() {
|
||||||
cd "$ROOT"
|
cd "$ROOT"
|
||||||
hook=".git/hooks/pre-commit"
|
hooks_dir="$(git rev-parse --git-common-dir)/hooks"
|
||||||
printf '#!/bin/sh\nset -e\nscript/precommit\n' > .git/hooks/pre-commit
|
mkdir -p "$hooks_dir"
|
||||||
chmod +x .git/hooks/pre-commit
|
hook="$hooks_dir/pre-commit"
|
||||||
|
printf '#!/bin/sh\nset -e\nscript/precommit\n' > "$hook"
|
||||||
|
chmod +x "$hook"
|
||||||
echo "pre-commit hook installed: runs script/precommit"
|
echo "pre-commit hook installed: runs script/precommit"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+19
-13
@@ -1,23 +1,29 @@
|
|||||||
#!/bin/sh
|
#!/bin/sh
|
||||||
# script/lint: run the linter. Linting is a phase of the Dockerfile and
|
# script/lint: run the linter. golangci-lint is never installed on a
|
||||||
# this builds that phase alone; the linter is never installed or run on
|
# host: it runs via docker only, one way, everywhere — this builds
|
||||||
# a developer host, where a shared result cache and a host-global lock
|
# Dockerfile.lint, which COPYs the repo into the digest-pinned
|
||||||
# make its answer untrustworthy.
|
# golangci-lint image and lints as a build step, so a successful build
|
||||||
|
# is a clean lint. The only prerequisite is a working docker. The gate
|
||||||
|
# steps make no network calls of their own, but Dockerfile.lint runs
|
||||||
|
# `go mod download` above them, so a cold cache does reach the network
|
||||||
|
# (as does pulling the pinned image); that layer stays cached, and once
|
||||||
|
# it is warm this runs offline until go.mod or go.sum changes.
|
||||||
#
|
#
|
||||||
# The phase is not the last stage in the file, so it is built only when
|
# CHECK_EPOCH is what makes the result mean anything. Without it docker
|
||||||
# --target names it. --no-cache because a cached lint layer is a lint
|
# serves the gate layers from cache on an unchanged tree and this exits
|
||||||
# that did not run. --output type=cacheonly writes no image, since
|
# 0 in well under a second having run no linter. The PID is in the value
|
||||||
# nothing uses one.
|
# as well as the epoch because two lint runs land inside the same second
|
||||||
|
# easily, and `date +%s` alone would cache the second one.
|
||||||
set -eu
|
set -eu
|
||||||
|
|
||||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||||
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
|
||||||
|
|
||||||
main() {
|
main() {
|
||||||
cd "$ROOT"
|
cd "$ROOT"
|
||||||
docker build --no-cache \
|
docker build \
|
||||||
--target lint \
|
--build-arg CHECK_EPOCH="$(date +%s)-$$" \
|
||||||
--output type=cacheonly .
|
-f Dockerfile.lint \
|
||||||
|
.
|
||||||
}
|
}
|
||||||
|
|
||||||
main "$@"
|
main "$@"
|
||||||
|
|||||||
+2
-2
@@ -1,13 +1,13 @@
|
|||||||
#!/bin/sh
|
#!/bin/sh
|
||||||
# script/precommit: run by the git pre-commit hook; fails the commit if
|
# script/precommit: run by the git pre-commit hook; fails the commit if
|
||||||
# checks fail. Our own extension to scripts-to-rule-them-all.
|
# checks fail. Our own extension to scripts-to-rule-them-all. Go extra:
|
||||||
|
# go mod tidy must be a no-op before the checks run.
|
||||||
set -eu
|
set -eu
|
||||||
|
|
||||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||||
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
||||||
|
|
||||||
main() {
|
main() {
|
||||||
# Go extra: go mod tidy must be a no-op before the checks run.
|
|
||||||
cd "$ROOT"
|
cd "$ROOT"
|
||||||
go mod tidy
|
go mod tidy
|
||||||
if ! git diff --exit-code -- go.mod go.sum; then
|
if ! git diff --exit-code -- go.mod go.sum; then
|
||||||
|
|||||||
+1
-1
@@ -1,6 +1,6 @@
|
|||||||
#!/bin/sh
|
#!/bin/sh
|
||||||
# script/setup: set up the repo for development after a fresh clone:
|
# script/setup: set up the repo for development after a fresh clone:
|
||||||
# installs dependencies and the git pre-commit hook.
|
# installs dependencies (script/bootstrap) and the git pre-commit hook.
|
||||||
set -eu
|
set -eu
|
||||||
|
|
||||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||||
|
|||||||
+8
-10
@@ -1,19 +1,17 @@
|
|||||||
#!/bin/sh
|
#!/bin/sh
|
||||||
# script/test: run the test suite. Testing is a phase of the Dockerfile
|
# script/test: run the test suite. Reruns verbosely on failure so CI
|
||||||
# and this builds that phase alone, on the same terms as script/lint:
|
# logs show which test failed.
|
||||||
# --target because a phase that is not the last stage is built only when
|
|
||||||
# named, and --no-cache because a cached test layer is a test that did
|
|
||||||
# not run. --output type=cacheonly writes no image, since nothing uses one.
|
|
||||||
set -eu
|
set -eu
|
||||||
|
|
||||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||||
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
|
||||||
|
|
||||||
main() {
|
main() {
|
||||||
cd "$ROOT"
|
cd "$ROOT"
|
||||||
docker build --no-cache \
|
go test -timeout 30s -cover ./... || {
|
||||||
--target test \
|
echo "--- Rerunning with -v for details ---"
|
||||||
--output type=cacheonly .
|
go test -timeout 30s -v ./...
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
main "$@"
|
main "$@"
|
||||||
|
|||||||
Executable
+84
@@ -0,0 +1,84 @@
|
|||||||
|
#!/bin/sh
|
||||||
|
# script/verify-lint-image-pin: fail unless the golangci-lint image
|
||||||
|
# referenced by Dockerfile.lint and the one referenced by the main
|
||||||
|
# Dockerfile's lint stage are the same image at the same digest. Our own
|
||||||
|
# extension to scripts-to-rule-them-all, not one of its entrypoints.
|
||||||
|
#
|
||||||
|
# The linter version is pinned in two independent files. That is the
|
||||||
|
# shape #42 turned into a build failure rather than tolerate: nothing
|
||||||
|
# else keeps the two in sync, and a bump applied to one file alone would
|
||||||
|
# leave `make lint` and the fail-fast lint stage of `make docker`
|
||||||
|
# linting the same tree against different rulesets, both green. This is
|
||||||
|
# the single guard that stops it, run as a gate in both files.
|
||||||
|
#
|
||||||
|
# It deliberately restates neither pin. A hardcoded expected digest here
|
||||||
|
# would be a third copy — one more thing to bump, and the same drift one
|
||||||
|
# file further out. It compares the two files to each other and knows
|
||||||
|
# nothing about which version is correct.
|
||||||
|
#
|
||||||
|
# A reference that cannot be read is a hard failure, not a skip: a
|
||||||
|
# comparison of two empty strings succeeds, which would turn this guard
|
||||||
|
# into exactly the unearned green it exists to prevent.
|
||||||
|
set -eu
|
||||||
|
|
||||||
|
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||||
|
|
||||||
|
LINT_DOCKERFILE="Dockerfile.lint"
|
||||||
|
MAIN_DOCKERFILE="Dockerfile"
|
||||||
|
|
||||||
|
# Echo the single golangci-lint image reference in the named Dockerfile.
|
||||||
|
# Scans every argument of every FROM instruction rather than assuming a
|
||||||
|
# field position, so `FROM --platform=... img AS stage` reads correctly.
|
||||||
|
# Exits non-zero, with a diagnosis, unless there is exactly one.
|
||||||
|
lint_image_ref() {
|
||||||
|
file="$1"
|
||||||
|
|
||||||
|
if [ ! -f "$file" ]; then
|
||||||
|
echo "verify-lint-image-pin: $file: not found" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
refs="$(
|
||||||
|
awk '
|
||||||
|
toupper($1) == "FROM" {
|
||||||
|
for (i = 2; i <= NF; i++) {
|
||||||
|
if ($i ~ /^golangci\/golangci-lint[:@]/) {
|
||||||
|
print $i
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
' "$file"
|
||||||
|
)"
|
||||||
|
|
||||||
|
count="$(printf '%s' "$refs" | grep -c . || true)"
|
||||||
|
if [ "$count" -ne 1 ]; then
|
||||||
|
echo "verify-lint-image-pin: $file: expected exactly one" \
|
||||||
|
"golangci/golangci-lint FROM reference, found $count" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
printf '%s\n' "$refs"
|
||||||
|
}
|
||||||
|
|
||||||
|
main() {
|
||||||
|
cd "$ROOT"
|
||||||
|
|
||||||
|
lint_ref="$(lint_image_ref "$LINT_DOCKERFILE")"
|
||||||
|
main_ref="$(lint_image_ref "$MAIN_DOCKERFILE")"
|
||||||
|
|
||||||
|
if [ "$lint_ref" != "$main_ref" ]; then
|
||||||
|
echo "verify-lint-image-pin: the linter image is pinned twice and" \
|
||||||
|
"the two pins disagree:" >&2
|
||||||
|
echo "verify-lint-image-pin: $LINT_DOCKERFILE: $lint_ref" >&2
|
||||||
|
echo "verify-lint-image-pin: $MAIN_DOCKERFILE: $main_ref" >&2
|
||||||
|
echo "verify-lint-image-pin: bump both FROM lines together so" \
|
||||||
|
"script/lint and the Dockerfile lint stage keep running the" \
|
||||||
|
"same linter" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "verify-lint-image-pin: $LINT_DOCKERFILE and $MAIN_DOCKERFILE" \
|
||||||
|
"agree on $lint_ref"
|
||||||
|
}
|
||||||
|
|
||||||
|
main "$@"
|
||||||
@@ -5,59 +5,49 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"crypto/sha256"
|
"crypto/sha256"
|
||||||
"fmt"
|
"fmt"
|
||||||
"io"
|
|
||||||
"os"
|
"os"
|
||||||
"slices"
|
"slices"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
|
|
||||||
// treeNode is one directory reconstructed from the record paths.
|
// fileSig is a file's duplicate signature; mtime is excluded.
|
||||||
|
type fileSig struct {
|
||||||
|
size int64
|
||||||
|
head string
|
||||||
|
tail string
|
||||||
|
content string
|
||||||
|
}
|
||||||
|
|
||||||
|
// treeNode is one directory reconstructed from the scan stream.
|
||||||
type treeNode struct {
|
type treeNode struct {
|
||||||
path string
|
path string
|
||||||
parent *treeNode
|
parent *treeNode
|
||||||
// entries holds the serialized child entries until the digest is
|
dirs map[string]*treeNode
|
||||||
// computed from them, and is then dropped.
|
files map[string]fileSig
|
||||||
entries []string
|
|
||||||
digest [sha256.Size]byte
|
digest [sha256.Size]byte
|
||||||
fileCount int64
|
fileCount int64
|
||||||
totalSize int64
|
totalSize int64
|
||||||
}
|
}
|
||||||
|
|
||||||
// runTrees implements the trees subcommand: it reads every record from
|
// runTrees implements the trees subcommand: it reads every record from
|
||||||
// the database in path order, reconstructs the directory hierarchy from
|
// the database, reconstructs the directory hierarchy from the record
|
||||||
// the record paths, computes a Merkle-style digest per directory, and
|
// paths, computes a Merkle-style digest per directory, and prints
|
||||||
// prints maximal duplicate-tree groups as TSV on stdout. It never
|
// maximal duplicate-tree groups as TSV on stdout. It never touches the
|
||||||
// touches the scanned filesystem; its only I/O is the database, stdout,
|
// scanned filesystem; its only I/O is the database, stdout, and
|
||||||
// and stderr. Any database problem, including a missing database, is
|
// stderr.
|
||||||
// fatal.
|
func runTrees(ctx context.Context) error {
|
||||||
func runTrees(ctx context.Context, stdout io.Writer) error {
|
recs, err := loadRecords(ctx)
|
||||||
dbPath := databasePath()
|
|
||||||
|
|
||||||
db, err := openReportDatabase(ctx, dbPath)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
defer func() { _ = db.Close() }()
|
super, allDirs := buildHierarchy(recs)
|
||||||
|
super.compute()
|
||||||
records := 0
|
|
||||||
tree := newTreeBuilder()
|
|
||||||
|
|
||||||
err = loadFileRows(ctx, db, func(r scanRec) {
|
|
||||||
records++
|
|
||||||
|
|
||||||
tree.add(r)
|
|
||||||
})
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("database %s: %w", dbPath, err)
|
|
||||||
}
|
|
||||||
|
|
||||||
super, allDirs := tree.finish()
|
|
||||||
|
|
||||||
dupes := collectTreeGroups(allDirs, super)
|
dupes := collectTreeGroups(allDirs, super)
|
||||||
|
|
||||||
out := bufio.NewWriterSize(stdout, ioBufSize)
|
out := bufio.NewWriterSize(os.Stdout, ioBufSize)
|
||||||
|
|
||||||
_, err = fmt.Fprintln(out, "first\tdupe\tfiles\tsize")
|
_, err = fmt.Fprintln(out, "first\tdupe\tfiles\tsize")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -72,8 +62,7 @@ func runTrees(ctx context.Context, stdout io.Writer) error {
|
|||||||
first := g[0]
|
first := g[0]
|
||||||
for _, n := range g[1:] {
|
for _, n := range g[1:] {
|
||||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\t%d\n",
|
_, err = fmt.Fprintf(out, "%s\t%s\t%d\t%d\n",
|
||||||
escapePath(first.path), escapePath(n.path),
|
first.path, n.path, first.fileCount, first.totalSize)
|
||||||
first.fileCount, first.totalSize)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("write stdout: %w", err)
|
return fmt.Errorf("write stdout: %w", err)
|
||||||
}
|
}
|
||||||
@@ -91,128 +80,65 @@ func runTrees(ctx context.Context, stdout io.Writer) error {
|
|||||||
fmt.Fprintf(os.Stderr,
|
fmt.Fprintf(os.Stderr,
|
||||||
"trees: %d records read, %d duplicate tree groups, %d dupe trees, "+
|
"trees: %d records read, %d duplicate tree groups, %d dupe trees, "+
|
||||||
"%s reclaimable\n",
|
"%s reclaimable\n",
|
||||||
records, len(dupes), dupeTrees, humanBytes(reclaimable))
|
len(recs), len(dupes), dupeTrees, humanBytes(reclaimable))
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// treeBuilder reconstructs the directory hierarchy from records added
|
// buildHierarchy reconstructs the directory hierarchy from the record
|
||||||
// in path order, under a synthetic super-root. Paths are split on "/";
|
// paths under a synthetic super-root. Paths are split on "/"; for
|
||||||
// for absolute paths the first component is empty, which becomes the
|
// absolute paths the first component is empty, which simply becomes a
|
||||||
// top-level directory with path "/". In path order all the paths under
|
// top-level node representing "/". It returns the super-root and every
|
||||||
// one directory come together, so a directory is complete once a path
|
// directory node created.
|
||||||
// outside it is added: its digest is computed then and its entries are
|
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
||||||
// dropped. Only the directories holding the latest path keep entries.
|
|
||||||
type treeBuilder struct {
|
|
||||||
super *treeNode
|
|
||||||
// open lists the directories holding the latest path, outermost
|
|
||||||
// first, starting with the super-root; names[i] is open[i]'s name.
|
|
||||||
open []*treeNode
|
|
||||||
names []string
|
|
||||||
// dirs lists every completed directory.
|
|
||||||
dirs []*treeNode
|
|
||||||
}
|
|
||||||
|
|
||||||
func newTreeBuilder() *treeBuilder {
|
|
||||||
super := &treeNode{}
|
super := &treeNode{}
|
||||||
|
|
||||||
return &treeBuilder{
|
var allDirs []*treeNode
|
||||||
super: super,
|
|
||||||
open: []*treeNode{super},
|
|
||||||
names: []string{""},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// add adds one record. Each record must come after the previous one in
|
for _, r := range recs {
|
||||||
// path order (byte order); otherwise a completed directory would be
|
|
||||||
// started again as a second directory with the same path.
|
|
||||||
func (b *treeBuilder) add(r scanRec) {
|
|
||||||
comps := strings.Split(r.path, "/")
|
comps := strings.Split(r.path, "/")
|
||||||
dirNames, name := comps[:len(comps)-1], comps[len(comps)-1]
|
|
||||||
|
|
||||||
// Keep the open directories that hold this path; complete the rest.
|
node := super
|
||||||
depth := 1
|
for _, c := range comps[:len(comps)-1] {
|
||||||
for depth < len(b.open) && depth <= len(dirNames) &&
|
child := node.dirs[c]
|
||||||
b.names[depth] == dirNames[depth-1] {
|
if child == nil {
|
||||||
depth++
|
childPath := c
|
||||||
|
if node != super {
|
||||||
|
childPath = node.path + "/" + c
|
||||||
}
|
}
|
||||||
|
|
||||||
b.closeTo(depth)
|
child = &treeNode{path: childPath, parent: node}
|
||||||
|
if node.dirs == nil {
|
||||||
for _, c := range dirNames[depth-1:] {
|
node.dirs = make(map[string]*treeNode)
|
||||||
b.openDir(c)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
dir := b.open[len(b.open)-1]
|
node.dirs[c] = child
|
||||||
dir.entries = append(dir.entries, fileEntry(name, r))
|
allDirs = append(allDirs, child)
|
||||||
dir.fileCount++
|
|
||||||
dir.totalSize += r.size
|
|
||||||
}
|
|
||||||
|
|
||||||
// openDir starts the directory called name inside the innermost open
|
|
||||||
// one.
|
|
||||||
func (b *treeBuilder) openDir(name string) {
|
|
||||||
parent := b.open[len(b.open)-1]
|
|
||||||
path := parent.path + "/" + name
|
|
||||||
|
|
||||||
// The root directory's path is "/", not empty, and its children's
|
|
||||||
// paths start with one slash, not two.
|
|
||||||
switch {
|
|
||||||
case parent == b.super && name == "":
|
|
||||||
path = "/"
|
|
||||||
case parent == b.super:
|
|
||||||
path = name
|
|
||||||
case parent.path == "/":
|
|
||||||
path = "/" + name
|
|
||||||
}
|
}
|
||||||
|
|
||||||
b.open = append(b.open, &treeNode{path: path, parent: parent})
|
node = child
|
||||||
b.names = append(b.names, name)
|
|
||||||
}
|
|
||||||
|
|
||||||
// closeTo completes the open directories after the first n, innermost
|
|
||||||
// first: each one's digest is computed and entered in its parent along
|
|
||||||
// with its totals.
|
|
||||||
func (b *treeBuilder) closeTo(n int) {
|
|
||||||
for len(b.open) > n {
|
|
||||||
last := len(b.open) - 1
|
|
||||||
dir, name := b.open[last], b.names[last]
|
|
||||||
b.open, b.names = b.open[:last], b.names[:last]
|
|
||||||
|
|
||||||
dir.computeDigest()
|
|
||||||
|
|
||||||
dir.parent.entries = append(dir.parent.entries,
|
|
||||||
"d\x00"+name+"\x00"+string(dir.digest[:]))
|
|
||||||
dir.parent.fileCount += dir.fileCount
|
|
||||||
dir.parent.totalSize += dir.totalSize
|
|
||||||
b.dirs = append(b.dirs, dir)
|
|
||||||
}
|
}
|
||||||
}
|
|
||||||
|
|
||||||
// finish completes every open directory and returns the super-root and
|
if node.files == nil {
|
||||||
// every directory.
|
node.files = make(map[string]fileSig)
|
||||||
func (b *treeBuilder) finish() (*treeNode, []*treeNode) {
|
}
|
||||||
b.closeTo(1)
|
|
||||||
|
|
||||||
return b.super, b.dirs
|
sig := fileSig{
|
||||||
}
|
size: r.size, head: r.head, tail: r.tail, content: r.content,
|
||||||
|
}
|
||||||
// fileEntry serializes a file child for its directory's digest: its
|
|
||||||
// name and its signature (size, head, tail, content); mtime is
|
|
||||||
// excluded.
|
|
||||||
func fileEntry(name string, r scanRec) string {
|
|
||||||
content := r.content
|
|
||||||
|
|
||||||
// A record without a content hash has unknown content (README
|
// A record without a content hash has unknown content (README
|
||||||
// "Database"): give it a signature no other file can share, so
|
// "Database"): give it a signature no other file can share, so
|
||||||
// trees containing it never compare equal. Real hashes are hex, so
|
// trees containing it never compare equal. Real hashes are
|
||||||
// the NUL-prefixed form cannot collide.
|
// hex, so the NUL-prefixed form cannot collide.
|
||||||
if content == "" {
|
if sig.content == "" {
|
||||||
content = "unhashed\x00" + r.path
|
sig.content = "unhashed\x00" + r.path
|
||||||
}
|
}
|
||||||
|
|
||||||
return "f\x00" + name + "\x00" + strconv.FormatInt(r.size, 10) +
|
node.files[comps[len(comps)-1]] = sig
|
||||||
"\x00" + r.head + "\x00" + r.tail + "\x00" + content
|
}
|
||||||
|
|
||||||
|
return super, allDirs
|
||||||
}
|
}
|
||||||
|
|
||||||
// collectTreeGroups groups directories by digest and returns every
|
// collectTreeGroups groups directories by digest and returns every
|
||||||
@@ -254,22 +180,38 @@ func collectTreeGroups(allDirs []*treeNode, super *treeNode) [][]*treeNode {
|
|||||||
return dupes
|
return dupes
|
||||||
}
|
}
|
||||||
|
|
||||||
// computeDigest sets n's digest and drops its entries. A directory's
|
// compute fills in digest, fileCount, and totalSize for n and all of
|
||||||
// digest is the SHA-256 of its child entries — files serialized with
|
// its descendants. A directory's digest is the SHA-256 of its child
|
||||||
// name and signature, subdirectories with name and recursive digest —
|
// entries — files serialized with name and signature, subdirectories
|
||||||
// sorted byte-lexicographically. Filenames cannot contain NUL or "/",
|
// with name and recursive digest — sorted byte-lexicographically.
|
||||||
// so NUL delimiters are unambiguous.
|
// Filenames cannot contain NUL or "/", so NUL delimiters are
|
||||||
func (n *treeNode) computeDigest() {
|
// unambiguous.
|
||||||
slices.Sort(n.entries)
|
func (n *treeNode) compute() {
|
||||||
|
entries := make([]string, 0, len(n.dirs)+len(n.files))
|
||||||
|
for name, sig := range n.files {
|
||||||
|
entries = append(entries,
|
||||||
|
"f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+
|
||||||
|
"\x00"+sig.head+"\x00"+sig.tail+"\x00"+sig.content)
|
||||||
|
n.fileCount++
|
||||||
|
n.totalSize += sig.size
|
||||||
|
}
|
||||||
|
|
||||||
|
for name, child := range n.dirs {
|
||||||
|
child.compute()
|
||||||
|
entries = append(entries, "d\x00"+name+"\x00"+string(child.digest[:]))
|
||||||
|
n.fileCount += child.fileCount
|
||||||
|
n.totalSize += child.totalSize
|
||||||
|
}
|
||||||
|
|
||||||
|
slices.Sort(entries)
|
||||||
|
|
||||||
h := sha256.New()
|
h := sha256.New()
|
||||||
for _, e := range n.entries {
|
for _, e := range entries {
|
||||||
h.Write([]byte(e))
|
h.Write([]byte(e))
|
||||||
h.Write([]byte{0})
|
h.Write([]byte{0})
|
||||||
}
|
}
|
||||||
|
|
||||||
copy(n.digest[:], h.Sum(nil))
|
copy(n.digest[:], h.Sum(nil))
|
||||||
n.entries = nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// suppressed reports whether a duplicate-tree group is non-maximal: its
|
// suppressed reports whether a duplicate-tree group is non-maximal: its
|
||||||
|
|||||||
+17
-129
@@ -1,8 +1,6 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
|
||||||
"database/sql"
|
|
||||||
"slices"
|
"slices"
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
@@ -31,36 +29,6 @@ func smokeTreeRecs() []scanRec {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// dbTree builds the directory hierarchy from the records in db the way
|
|
||||||
// trees does, and returns the super-root and every directory.
|
|
||||||
func dbTree(t *testing.T, db *sql.DB) (*treeNode, []*treeNode) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
tree := newTreeBuilder()
|
|
||||||
|
|
||||||
err := loadFileRows(t.Context(), db, tree.add)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return tree.finish()
|
|
||||||
}
|
|
||||||
|
|
||||||
// treeOf writes recs into a fresh database and builds the directory
|
|
||||||
// hierarchy from it the way trees does.
|
|
||||||
func treeOf(t *testing.T, recs []scanRec) (*treeNode, []*treeNode) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
db := openTestDB(t)
|
|
||||||
|
|
||||||
err := applyChanges(t.Context(), db, recs, nil, nil)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return dbTree(t, db)
|
|
||||||
}
|
|
||||||
|
|
||||||
// nodeByPath finds the directory node with the given path.
|
// nodeByPath finds the directory node with the given path.
|
||||||
func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode {
|
func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
@@ -91,10 +59,11 @@ func groupPaths(groups [][]*treeNode) [][]string {
|
|||||||
return out
|
return out
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestTreeCounts(t *testing.T) {
|
func TestBuildHierarchyCounts(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
_, dirs := treeOf(t, smokeTreeRecs())
|
super, dirs := buildHierarchy(smokeTreeRecs())
|
||||||
|
super.compute()
|
||||||
|
|
||||||
d := nodeByPath(t, dirs, "/d")
|
d := nodeByPath(t, dirs, "/d")
|
||||||
if d.fileCount != 6 || d.totalSize != 9300 {
|
if d.fileCount != 6 || d.totalSize != 9300 {
|
||||||
@@ -115,98 +84,11 @@ func TestTreeCounts(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestTreeRootPath(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
// The root directory's path is "/", never empty, and its
|
|
||||||
// children's paths start with a single slash.
|
|
||||||
_, dirs := treeOf(t, []scanRec{{path: "/f"}, {path: "/srv/g"}})
|
|
||||||
|
|
||||||
got := make([]string, 0, len(dirs))
|
|
||||||
for _, d := range dirs {
|
|
||||||
got = append(got, d.path)
|
|
||||||
}
|
|
||||||
|
|
||||||
slices.Sort(got)
|
|
||||||
|
|
||||||
want := []string{"/", "/srv"}
|
|
||||||
if !slices.Equal(got, want) {
|
|
||||||
t.Fatalf("directory paths = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestTreeNamesSortingBeforeSlash(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
// In path order "/a/b-x/f" and "/a/b.txt" come between the file
|
|
||||||
// "/a/b" and "/a/b/f", because "-" and "." sort before "/". Each
|
|
||||||
// directory must still be built once, whole, so /a matches /c.
|
|
||||||
recs := make([]scanRec, 0, 8)
|
|
||||||
|
|
||||||
for _, top := range []string{"/a", "/c"} {
|
|
||||||
for _, p := range []string{"/b", "/b-x/f", "/b.txt", "/b/f"} {
|
|
||||||
content := "c"
|
|
||||||
if p == "/b-x/f" {
|
|
||||||
content = "other"
|
|
||||||
}
|
|
||||||
|
|
||||||
recs = append(recs, scanRec{
|
|
||||||
size: 1, head: "h", tail: "t", content: content, path: top + p,
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
super, dirs := treeOf(t, recs)
|
|
||||||
|
|
||||||
got := make([]string, 0, len(dirs))
|
|
||||||
for _, d := range dirs {
|
|
||||||
got = append(got, d.path)
|
|
||||||
}
|
|
||||||
|
|
||||||
slices.Sort(got)
|
|
||||||
|
|
||||||
want := []string{"/", "/a", "/a/b", "/a/b-x", "/c", "/c/b", "/c/b-x"}
|
|
||||||
if !slices.Equal(got, want) {
|
|
||||||
t.Fatalf("directory paths = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
groups := collectTreeGroups(dirs, super)
|
|
||||||
|
|
||||||
gotGroups := groupPaths(groups)
|
|
||||||
wantGroups := [][]string{{"/a", "/c"}}
|
|
||||||
|
|
||||||
if !slices.EqualFunc(gotGroups, wantGroups, slices.Equal) {
|
|
||||||
t.Fatalf("groups = %v, want %v", gotGroups, wantGroups)
|
|
||||||
}
|
|
||||||
|
|
||||||
if groups[0][0].fileCount != 4 || groups[0][0].totalSize != 4 {
|
|
||||||
t.Errorf("group totals: %d files %d bytes, want 4 4",
|
|
||||||
groups[0][0].fileCount, groups[0][0].totalSize)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunTreesEscapesPaths(t *testing.T) {
|
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{cmdTrees}, &stdout, &stderr)
|
|
||||||
if code != exitOK {
|
|
||||||
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
|
||||||
code, exitOK, stderr.String())
|
|
||||||
}
|
|
||||||
|
|
||||||
want := "first\tdupe\tfiles\tsize\n" +
|
|
||||||
`/d/\tone\ntwo\rthree\\four` + "\t/d/A\t1\t5\n"
|
|
||||||
if got := stdout.String(); got != want {
|
|
||||||
t.Errorf("stdout = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestTreeDigests(t *testing.T) {
|
func TestTreeDigests(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
_, dirs := treeOf(t, smokeTreeRecs())
|
super, dirs := buildHierarchy(smokeTreeRecs())
|
||||||
|
super.compute()
|
||||||
|
|
||||||
t1 := nodeByPath(t, dirs, "/d/t1")
|
t1 := nodeByPath(t, dirs, "/d/t1")
|
||||||
t2 := nodeByPath(t, dirs, "/d/t2")
|
t2 := nodeByPath(t, dirs, "/d/t2")
|
||||||
@@ -239,7 +121,8 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
|
|||||||
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
|
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
_, dirs := treeOf(t, recs)
|
super, dirs := buildHierarchy(recs)
|
||||||
|
super.compute()
|
||||||
|
|
||||||
a := nodeByPath(t, dirs, "/r/a")
|
a := nodeByPath(t, dirs, "/r/a")
|
||||||
b := nodeByPath(t, dirs, "/r/b")
|
b := nodeByPath(t, dirs, "/r/b")
|
||||||
@@ -252,7 +135,8 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
|
|||||||
func TestCollectTreeGroupsMaximal(t *testing.T) {
|
func TestCollectTreeGroupsMaximal(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
super, dirs := treeOf(t, smokeTreeRecs())
|
super, dirs := buildHierarchy(smokeTreeRecs())
|
||||||
|
super.compute()
|
||||||
|
|
||||||
groups := collectTreeGroups(dirs, super)
|
groups := collectTreeGroups(dirs, super)
|
||||||
|
|
||||||
@@ -276,14 +160,16 @@ func TestCollectTreeGroupsDeterministic(t *testing.T) {
|
|||||||
|
|
||||||
recs := smokeTreeRecs()
|
recs := smokeTreeRecs()
|
||||||
|
|
||||||
super, dirs := treeOf(t, recs)
|
super, dirs := buildHierarchy(recs)
|
||||||
|
super.compute()
|
||||||
|
|
||||||
forward := groupPaths(collectTreeGroups(dirs, super))
|
forward := groupPaths(collectTreeGroups(dirs, super))
|
||||||
|
|
||||||
reversed := slices.Clone(recs)
|
reversed := slices.Clone(recs)
|
||||||
slices.Reverse(reversed)
|
slices.Reverse(reversed)
|
||||||
|
|
||||||
superR, dirsR := treeOf(t, reversed)
|
superR, dirsR := buildHierarchy(reversed)
|
||||||
|
superR.compute()
|
||||||
|
|
||||||
backward := groupPaths(collectTreeGroups(dirsR, superR))
|
backward := groupPaths(collectTreeGroups(dirsR, superR))
|
||||||
if !slices.EqualFunc(forward, backward, slices.Equal) {
|
if !slices.EqualFunc(forward, backward, slices.Equal) {
|
||||||
@@ -302,7 +188,8 @@ func TestCollectTreeGroupsSiblings(t *testing.T) {
|
|||||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
|
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := treeOf(t, recs)
|
super, dirs := buildHierarchy(recs)
|
||||||
|
super.compute()
|
||||||
|
|
||||||
got := groupPaths(collectTreeGroups(dirs, super))
|
got := groupPaths(collectTreeGroups(dirs, super))
|
||||||
|
|
||||||
@@ -324,7 +211,8 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) {
|
|||||||
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
|
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := treeOf(t, recs)
|
super, dirs := buildHierarchy(recs)
|
||||||
|
super.compute()
|
||||||
|
|
||||||
got := groupPaths(collectTreeGroups(dirs, super))
|
got := groupPaths(collectTreeGroups(dirs, super))
|
||||||
|
|
||||||
|
|||||||
@@ -1,8 +0,0 @@
|
|||||||
# THIS IS AN AUTOGENERATED FILE. DO NOT EDIT THIS FILE DIRECTLY.
|
|
||||||
# yarn lockfile v1
|
|
||||||
|
|
||||||
|
|
||||||
prettier@3.8.1:
|
|
||||||
version "3.8.1"
|
|
||||||
resolved "https://registry.yarnpkg.com/prettier/-/prettier-3.8.1.tgz#edf48977cf991558f4fcbd8a3ba6015ba2a3a173"
|
|
||||||
integrity sha512-UOnG6LftzbdaHZcKoPFtOcCKztrQ57WkHDeRD9t/PTQtmT0NHSeWWepj6pS0z/N7+08BHFDQVUrfmfMRcZwbMg==
|
|
||||||
Reference in New Issue
Block a user