Compare commits
1
Commits
next
..
7cbb2d7f20
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7cbb2d7f20 |
+2
-66
@@ -10,20 +10,14 @@ run:
|
|||||||
|
|
||||||
linters:
|
linters:
|
||||||
default: all
|
default: all
|
||||||
enable:
|
|
||||||
# Successor to the deprecated gomodguard. Named explicitly, rather than
|
|
||||||
# left to `default: all`, because it carries the module policy below.
|
|
||||||
- gomodguard_v2
|
|
||||||
disable:
|
disable:
|
||||||
# Genuinely incompatible with project patterns
|
# Genuinely incompatible with project patterns
|
||||||
- exhaustruct # Requires all struct fields
|
- exhaustruct # Requires all struct fields
|
||||||
|
- depguard # Dependency allow/block lists
|
||||||
- godot # Requires comments to end with periods
|
- godot # Requires comments to end with periods
|
||||||
|
- wsl # Deprecated, replaced by wsl_v5
|
||||||
- wrapcheck # Too verbose for internal packages
|
- wrapcheck # Too verbose for internal packages
|
||||||
- varnamelen # Short names like db, id are idiomatic Go
|
- varnamelen # Short names like db, id are idiomatic Go
|
||||||
# Deprecated: the warning is attached to the old name, so it is
|
|
||||||
# silenced by disabling that name, not by enabling the successor.
|
|
||||||
- wsl # Deprecated, replaced by wsl_v5
|
|
||||||
- gomodguard # Deprecated, replaced by gomodguard_v2
|
|
||||||
settings:
|
settings:
|
||||||
lll:
|
lll:
|
||||||
line-length: 88
|
line-length: 88
|
||||||
@@ -34,64 +28,6 @@ linters:
|
|||||||
max-complexity: 15
|
max-complexity: 15
|
||||||
dupl:
|
dupl:
|
||||||
threshold: 100
|
threshold: 100
|
||||||
depguard:
|
|
||||||
# Test-support code must not be compiled into the shipped binary. A
|
|
||||||
# test-support package exists to hand a test privileges the program
|
|
||||||
# itself must never have, so a file that is not a test must not import
|
|
||||||
# one. Test files, and the files inside a package whose directory name
|
|
||||||
# ends in `test`, are where that code belongs, and are exempt.
|
|
||||||
#
|
|
||||||
# The deny list below is the one part of this file a repository is
|
|
||||||
# expected to extend, and the only part it may. depguard matches an
|
|
||||||
# import path against a list of prefixes, so it cannot be told "any path
|
|
||||||
# whose last segment ends in test"; a repository's own test-support
|
|
||||||
# packages have to be named here one at a time, by full import path,
|
|
||||||
# under a module path that differs from repository to repository. Add
|
|
||||||
# them; change nothing else.
|
|
||||||
rules:
|
|
||||||
test-support:
|
|
||||||
list-mode: lax
|
|
||||||
files:
|
|
||||||
- "$all"
|
|
||||||
- "!$test"
|
|
||||||
- "!**/*test/**"
|
|
||||||
deny:
|
|
||||||
- pkg: net/http/httptest
|
|
||||||
desc: >-
|
|
||||||
Test-support code belongs in test files and in packages whose
|
|
||||||
directory name ends in test, not in the shipped binary.
|
|
||||||
# Only decisions already recorded in the Go package defaults are
|
|
||||||
# listed here. Every entry matches the module path exactly.
|
|
||||||
gomodguard_v2:
|
|
||||||
blocked:
|
|
||||||
- module: github.com/rs/zerolog
|
|
||||||
recommendations:
|
|
||||||
- log/slog
|
|
||||||
reason: "Structured logging is stdlib log/slog."
|
|
||||||
# One entry per pre-fork module path, because the later releases
|
|
||||||
# are separate paths. A prefix match would be shorter but would
|
|
||||||
# also reach github.com/go-redis/redismock, the test double for
|
|
||||||
# the successor these entries recommend.
|
|
||||||
- module: github.com/go-redis/redis
|
|
||||||
recommendations:
|
|
||||||
- github.com/redis/go-redis/v9
|
|
||||||
reason: "Pre-fork module; use the maintained go-redis v9."
|
|
||||||
- module: github.com/go-redis/redis/v7
|
|
||||||
recommendations:
|
|
||||||
- github.com/redis/go-redis/v9
|
|
||||||
reason: "Pre-fork module; use the maintained go-redis v9."
|
|
||||||
- module: github.com/go-redis/redis/v8
|
|
||||||
recommendations:
|
|
||||||
- github.com/redis/go-redis/v9
|
|
||||||
reason: "Pre-fork module; use the maintained go-redis v9."
|
|
||||||
- module: github.com/sergi/go-diff
|
|
||||||
recommendations:
|
|
||||||
- github.com/aymanbagabas/go-udiff
|
|
||||||
reason: "No unified diff output; use go-udiff."
|
|
||||||
- module: github.com/hexops/gotextdiff
|
|
||||||
recommendations:
|
|
||||||
- github.com/aymanbagabas/go-udiff
|
|
||||||
reason: "Unmaintained fork; use go-udiff."
|
|
||||||
|
|
||||||
issues:
|
issues:
|
||||||
max-issues-per-linter: 0
|
max-issues-per-linter: 0
|
||||||
|
|||||||
+8
-26
@@ -51,21 +51,14 @@ RUN echo "gate lint, epoch ${CHECK_EPOCH}" && \
|
|||||||
FROM golang@sha256:56961d79ea8129efddcc0b8643fd8a5416b4e6228cfd477e3fd61deb2672c587 AS builder
|
FROM golang@sha256:56961d79ea8129efddcc0b8643fd8a5416b4e6228cfd477e3fd61deb2672c587 AS builder
|
||||||
|
|
||||||
# We never build or run as root. Create an unprivileged user and point
|
# We never build or run as root. Create an unprivileged user and point
|
||||||
# HOME and the build cache at its home so go build and go test can write
|
# HOME and the Go caches at its home so go build and go test can write
|
||||||
# it when we drop to it below. $GOPATH/bin is deliberately not on PATH:
|
# their caches when we drop to it below. $GOPATH/bin is deliberately not
|
||||||
# script/bootstrap no longer `go install`s anything (the linter runs
|
# on PATH: script/bootstrap no longer `go install`s anything (the linter
|
||||||
# from a pinned image, never from a host install), so nothing lands
|
# runs from a pinned image, never from a host install), so nothing lands
|
||||||
# there and adding it would only widen what this image resolves.
|
# there and adding it would only widen what this image resolves.
|
||||||
#
|
|
||||||
# The module cache is kept outside that home, at the base image's
|
|
||||||
# default /go/pkg/mod, and belongs to root: script/bootstrap fills it as
|
|
||||||
# root. Do not move it into the home and hand it over with `chown -R`:
|
|
||||||
# that walks every file in it, which took from about 80 s to over ten
|
|
||||||
# minutes on a shared host, depending on load.
|
|
||||||
RUN adduser -D -u 1000 builder
|
RUN adduser -D -u 1000 builder
|
||||||
ENV HOME=/home/builder
|
ENV HOME=/home/builder
|
||||||
ENV GOPATH=/home/builder/go
|
ENV GOPATH=/home/builder/go
|
||||||
ENV GOMODCACHE=/go/pkg/mod
|
|
||||||
ENV GOCACHE=/home/builder/.cache/go-build
|
ENV GOCACHE=/home/builder/.cache/go-build
|
||||||
|
|
||||||
WORKDIR /src
|
WORKDIR /src
|
||||||
@@ -90,22 +83,11 @@ COPY script/ script/
|
|||||||
COPY go.mod go.sum ./
|
COPY go.mod go.sum ./
|
||||||
RUN script/bootstrap
|
RUN script/bootstrap
|
||||||
|
|
||||||
# Hand builder only what it writes to, without walking the module cache.
|
COPY . .
|
||||||
# This layer stays cached with bootstrap.
|
|
||||||
# - /src itself: make build writes the binary into it, and git refuses
|
|
||||||
# a repository whose top directory belongs to another user.
|
|
||||||
# - the module cache's cache/download directory itself, not what is in
|
|
||||||
# it: Go only reads the downloaded modules, but make build saves its
|
|
||||||
# lookup of this module's own version from git there, in a new
|
|
||||||
# directory named after the module path.
|
|
||||||
# - builder's home: the go commands bootstrap ran as root left Go's
|
|
||||||
# telemetry files there, a few small files.
|
|
||||||
RUN chown builder:builder /src /go/pkg/mod/cache/download && \
|
|
||||||
chown -R builder:builder /home/builder
|
|
||||||
|
|
||||||
# The sources are handed to builder as they are copied, so no layer has
|
# Hand the sources and caches to the unprivileged user, then drop root
|
||||||
# to walk them. Then drop root before running any checks or builds.
|
# before running any checks or builds.
|
||||||
COPY --chown=builder:builder . .
|
RUN chown -R builder:builder /src /home/builder
|
||||||
USER builder
|
USER builder
|
||||||
|
|
||||||
# Fail the build unless the branch is green. Runs as non-root so the
|
# Fail the build unless the branch is green. Runs as non-root so the
|
||||||
|
|||||||
@@ -51,122 +51,6 @@ daily `sfdupes scan` cron job, with the reporting commands run
|
|||||||
interactively whenever needed; their results are as fresh as the last
|
interactively whenever needed; their results are as fresh as the last
|
||||||
completed scan.
|
completed scan.
|
||||||
|
|
||||||
### Install
|
|
||||||
|
|
||||||
With Go installed, this builds and installs the current `main` branch:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
go install sneak.berlin/go/sfdupes@main
|
|
||||||
```
|
|
||||||
|
|
||||||
The binary goes to `$(go env GOPATH)/bin`, or to `$GOBIN` when that is
|
|
||||||
set. A binary installed this way reports its version as `dev`; one
|
|
||||||
built from a clone or into the Docker image carries the git tag or
|
|
||||||
commit it was built from.
|
|
||||||
|
|
||||||
From a clone, `make build` writes the binary to `./sfdupes`:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
git clone https://git.eeqj.de/sneak/sfdupes.git
|
|
||||||
cd sfdupes
|
|
||||||
make build
|
|
||||||
```
|
|
||||||
|
|
||||||
Copy the binary to `/usr/local/bin` for the cron job below.
|
|
||||||
|
|
||||||
`make docker` builds the Docker image, tagged `sfdupes`, after running
|
|
||||||
the tests and the linter (see "Build"). The image runs `sfdupes` as
|
|
||||||
root with the database at its default path, so a bind mount of
|
|
||||||
`/var/lib/sfdupes` keeps the database between runs. Mount the scanned
|
|
||||||
tree at the same path inside the container as on the host; read-only
|
|
||||||
is enough. The database records paths as the container sees them, so
|
|
||||||
the reports then name the host's paths.
|
|
||||||
|
|
||||||
```sh
|
|
||||||
make docker
|
|
||||||
docker run --rm -v /srv:/srv:ro -v /var/lib/sfdupes:/var/lib/sfdupes \
|
|
||||||
sfdupes scan /srv
|
|
||||||
docker run --rm -v /var/lib/sfdupes:/var/lib/sfdupes sfdupes report > dupes.tsv
|
|
||||||
```
|
|
||||||
|
|
||||||
### Daily scan from cron
|
|
||||||
|
|
||||||
Run `scan` as root, so that it can read every file: a path it cannot
|
|
||||||
read is skipped with a warning and loses its database record (see
|
|
||||||
"Rules for the walk"). As a file `/etc/cron.d/sfdupes`:
|
|
||||||
|
|
||||||
```
|
|
||||||
30 3 * * * root /usr/local/bin/sfdupes scan /srv 2>>/var/log/sfdupes.log || tail -n 3 /var/log/sfdupes.log
|
|
||||||
```
|
|
||||||
|
|
||||||
- The database is `/var/lib/sfdupes/db.sqlite`, created with its
|
|
||||||
directory by the first scan. To keep it elsewhere, set
|
|
||||||
`SFDUPES_DATABASE=/path/to/db.sqlite` before the command on the same
|
|
||||||
line.
|
|
||||||
- `scan` writes nothing to stdout. Its stderr, appended here to
|
|
||||||
`/var/log/sfdupes.log`, holds a plain progress line as each phase
|
|
||||||
starts and then at most every 5 seconds, a warning for each path it
|
|
||||||
skips, and the summary line (see "Progress" and "`scan` mode"). The
|
|
||||||
log grows with every scan; rotate it like any other.
|
|
||||||
- Skipped paths do not fail a scan: it still exits 0, and cron sends
|
|
||||||
nothing. A scan that fails, or is stopped by `SIGINT` or `SIGTERM`,
|
|
||||||
exits 1 with the reason among the last lines of the log; `tail`
|
|
||||||
prints them, and cron mails them to root if the host can send mail.
|
|
||||||
- A scan still running when the next one starts carries on. The new
|
|
||||||
one fails at once, and the lines cron mails include
|
|
||||||
`sfdupes: another scan is running (lock held on /var/lib/sfdupes/db.sqlite.lock)`.
|
|
||||||
- `report` and `trees` need only read access to the database (see
|
|
||||||
"Database"). Under the usual umask of `022` the first scan creates
|
|
||||||
it readable by every user, so an unprivileged user can run them
|
|
||||||
against root's database.
|
|
||||||
|
|
||||||
### Reading the reports
|
|
||||||
|
|
||||||
Each row of `report` names two copies of one file, and each row of
|
|
||||||
`trees` two copies of one directory tree (see "Report output format"
|
|
||||||
and "Trees output format"). In a group of copies, the path that sorts
|
|
||||||
first byte by byte is `first` and every other path is a `dupe` of it.
|
|
||||||
`first` says nothing about which copy is the original or the oldest;
|
|
||||||
which copy to keep is your choice.
|
|
||||||
|
|
||||||
A row is a candidate, not proof:
|
|
||||||
|
|
||||||
- The reports read only the database, so they show the files as of
|
|
||||||
the last scan; a file may have changed or gone since.
|
|
||||||
- A file of 50 MiB or more is compared only on samples of its content
|
|
||||||
(see "Duplicate detection").
|
|
||||||
- Paths that are hard links to one file are listed as duplicates, but
|
|
||||||
they share their data, so removing one frees nothing.
|
|
||||||
|
|
||||||
Compare a pair byte for byte before removing either copy. For the row
|
|
||||||
`/srv/a/big.iso`, `/srv/b/big-copy.iso`, `4294967296`:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
cmp /srv/a/big.iso /srv/b/big-copy.iso && echo identical
|
|
||||||
[ /srv/a/big.iso -ef /srv/b/big-copy.iso ] && echo "hard links"
|
|
||||||
```
|
|
||||||
|
|
||||||
`cmp` prints nothing and exits 0 only when every byte matches, and
|
|
||||||
otherwise reports where the files differ. The second line prints
|
|
||||||
`hard links` when the two paths are the same file, so removing either
|
|
||||||
frees nothing.
|
|
||||||
|
|
||||||
A path holding a backslash, tab, newline or carriage return is escaped
|
|
||||||
in the reports (see "Report output format"). Undo the escapes before
|
|
||||||
using it. `printf '%b'` does exactly that, because every backslash in
|
|
||||||
an escaped path starts one of the four escapes. Command substitution
|
|
||||||
drops trailing newlines, so print an `x` after the path and remove it
|
|
||||||
afterwards, or a path that ends in a newline names a different file:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
p="$(printf '%bx' '/srv/a/tab\tname.txt')"; p="${p%x}"
|
|
||||||
cmp "$p" /srv/b/tab-copy.txt
|
|
||||||
```
|
|
||||||
|
|
||||||
Check a `trees` row with `diff -r`, which compares the two trees file
|
|
||||||
by file and also names anything present in only one of them, such as
|
|
||||||
an empty directory or a symlink, which `trees` does not see.
|
|
||||||
|
|
||||||
## Rationale
|
## Rationale
|
||||||
|
|
||||||
Duplicate finders that hash entire files do not scale to the target
|
Duplicate finders that hash entire files do not scale to the target
|
||||||
@@ -228,8 +112,8 @@ Goals, in order:
|
|||||||
again. `scan` is designed to be cronned; the reports run at any
|
again. `scan` is designed to be cronned; the reports run at any
|
||||||
time against the last completed scan.
|
time against the last completed scan.
|
||||||
4. **Clean stream separation.** Everything on stdout is machine-readable
|
4. **Clean stream separation.** Everything on stdout is machine-readable
|
||||||
data. All progress, warnings, summaries, and help and usage text go
|
data. All progress, warnings, and summaries go to stderr. Never mix
|
||||||
to stderr. Never mix them.
|
them.
|
||||||
|
|
||||||
### Constraints
|
### Constraints
|
||||||
|
|
||||||
@@ -265,27 +149,15 @@ Three subcommands, all implemented:
|
|||||||
sfdupes scan [--workers N] [-x] PATH...
|
sfdupes scan [--workers N] [-x] PATH...
|
||||||
sfdupes report > dupes.tsv
|
sfdupes report > dupes.tsv
|
||||||
sfdupes trees > dupetrees.tsv
|
sfdupes trees > dupetrees.tsv
|
||||||
sfdupes --version
|
|
||||||
sfdupes [command] --help
|
|
||||||
```
|
```
|
||||||
|
|
||||||
`--workers N` sets the size of each `scan` worker pool (default: the
|
|
||||||
number of CPUs), and `-x` (`--one-file-system`) keeps the walk of each
|
|
||||||
operand on that operand's filesystem; both are described under
|
|
||||||
"`scan` mode". `sfdupes --version` (or `-v`) prints one line,
|
|
||||||
`sfdupes VERSION`, to stdout and exits 0, writing nothing to stderr.
|
|
||||||
`-h` or `--help`, alone or after a subcommand, prints the help text to
|
|
||||||
stderr and exits 0, writing nothing to stdout.
|
|
||||||
|
|
||||||
### Database
|
### Database
|
||||||
|
|
||||||
All three subcommands operate on a single SQLite database file:
|
All three subcommands operate on a single SQLite database file:
|
||||||
|
|
||||||
- Location: the value of the `SFDUPES_DATABASE` environment variable
|
- Location: the value of the `SFDUPES_DATABASE` environment variable
|
||||||
when set and non-empty, otherwise `/var/lib/sfdupes/db.sqlite`.
|
when set and non-empty, otherwise `/var/lib/sfdupes/db.sqlite`.
|
||||||
There is no command-line flag. The path names the file exactly,
|
There is no command-line flag.
|
||||||
whatever characters it holds (`?`, `#` and `%` included); a
|
|
||||||
relative path is relative to the working directory.
|
|
||||||
- `scan` creates the database (and its parent directory) on first
|
- `scan` creates the database (and its parent directory) on first
|
||||||
use. `report` and `trees` require an existing database; a missing
|
use. `report` and `trees` require an existing database; a missing
|
||||||
database file is a fatal error (exit 1) telling the user to run
|
database file is a fatal error (exit 1) telling the user to run
|
||||||
@@ -312,9 +184,8 @@ All three subcommands operate on a single SQLite database file:
|
|||||||
(keeping the WAL small and letting concurrent reports observe
|
(keeping the WAL small and letting concurrent reports observe
|
||||||
progress), so a report may see a scan's changes partially applied,
|
progress), so a report may see a scan's changes partially applied,
|
||||||
and a scan that dies partway leaves a valid database holding
|
and a scan that dies partway leaves a valid database holding
|
||||||
every batch committed so far (an interrupted scan also commits the
|
everything hashed so far; the next scan skips those records and
|
||||||
batch in progress, see "Error handling and exit codes"); the next
|
converges toward the filesystem.
|
||||||
scan skips those records and converges toward the filesystem.
|
|
||||||
- `scan` switches the database back to rollback-journal mode when it
|
- `scan` switches the database back to rollback-journal mode when it
|
||||||
closes it, so between scans the database file alone holds the whole
|
closes it, so between scans the database file alone holds the whole
|
||||||
database. Each switch needs the database to itself: a `scan` that
|
database. Each switch needs the database to itself: a `scan` that
|
||||||
@@ -331,12 +202,7 @@ All three subcommands operate on a single SQLite database file:
|
|||||||
`-shm` files beside it, which SQLite creates with the database
|
`-shm` files beside it, which SQLite creates with the database
|
||||||
file's permissions.
|
file's permissions.
|
||||||
- Schema (`PRAGMA user_version` is the schema version, currently 1; a
|
- Schema (`PRAGMA user_version` is the schema version, currently 1; a
|
||||||
database with any other version is a fatal error. `scan` creates
|
database with any other version is a fatal error):
|
||||||
the schema and sets the version in one transaction, so a first scan
|
|
||||||
stopped while doing so leaves an empty database the next scan sets
|
|
||||||
up. A database at version 0 that already has a `files` table was
|
|
||||||
therefore not made by sfdupes; every subcommand refuses it with an
|
|
||||||
error telling the user to remove the file and rescan):
|
|
||||||
|
|
||||||
```sql
|
```sql
|
||||||
CREATE TABLE files (
|
CREATE TABLE files (
|
||||||
@@ -568,9 +434,7 @@ Rules for the walk:
|
|||||||
Concurrency: the walk phase (which also stats files), the hash phase,
|
Concurrency: the walk phase (which also stats files), the hash phase,
|
||||||
and the content phase each use a worker pool of `--workers` workers
|
and the content phase each use a worker pool of `--workers` workers
|
||||||
(default `runtime.NumCPU()`); the walk parallelizes across
|
(default `runtime.NumCPU()`); the walk parallelizes across
|
||||||
directories, hashing across files. `--workers` must be at least 1: a
|
directories, hashing across files. All three phases are seek-bound on
|
||||||
smaller value is a usage error, reported in one line on stderr with
|
|
||||||
exit 2 before anything is scanned. All three phases are seek-bound on
|
|
||||||
spinning disks, so raising `--workers` well past the core count can
|
spinning disks, so raising `--workers` well past the core count can
|
||||||
help on pools with many spindles. The main goroutine owns
|
help on pools with many spindles. The main goroutine owns
|
||||||
partitioning, database writes, and progress rendering; progress
|
partitioning, database writes, and progress rendering; progress
|
||||||
@@ -752,8 +616,6 @@ Additional requirements:
|
|||||||
waits for its next item.
|
waits for its next item.
|
||||||
- A warning printed during a phase always lands on a line of its own,
|
- A warning printed during a phase always lands on a line of its own,
|
||||||
never inside the progress display.
|
never inside the progress display.
|
||||||
- A bar whose phase stops short of its total, as an interrupted one
|
|
||||||
does, is left as last drawn rather than filled up.
|
|
||||||
- `report` and `trees` modes need no progress display, only their
|
- `report` and `trees` modes need no progress display, only their
|
||||||
stderr summaries.
|
stderr summaries.
|
||||||
|
|
||||||
@@ -763,11 +625,9 @@ Additional requirements:
|
|||||||
- `1`: fatal error (e.g., a `PATH` operand does not exist, another
|
- `1`: fatal error (e.g., a `PATH` operand does not exist, another
|
||||||
`scan` is already running against the same database, the database
|
`scan` is already running against the same database, the database
|
||||||
cannot be created/opened/read/written, a missing database for
|
cannot be created/opened/read/written, a missing database for
|
||||||
`report`/`trees`, stdout write failure), or a `scan` stopped by
|
`report`/`trees`, stdout write failure).
|
||||||
`SIGINT` or `SIGTERM` (see below).
|
- `2`: usage error (including `scan` with no `PATH` operand and
|
||||||
- `2`: usage error (including `scan` with no `PATH` operand, `scan`
|
`report`/`trees` with any positional argument).
|
||||||
with `--workers` below 1, and `report`/`trees` with any positional
|
|
||||||
argument).
|
|
||||||
|
|
||||||
A stdout write failure, such as a full disk, is reported in one line on
|
A stdout write failure, such as a full disk, is reported in one line on
|
||||||
stderr and exits 1. Two cases never reach sfdupes as a failed write:
|
stderr and exits 1. Two cases never reach sfdupes as a failed write:
|
||||||
@@ -781,24 +641,6 @@ stderr and exits 1. Two cases never reach sfdupes as a failed write:
|
|||||||
the output is discarded and the run succeeds, as with
|
the output is discarded and the run succeeds, as with
|
||||||
`> /dev/null`.
|
`> /dev/null`.
|
||||||
|
|
||||||
`scan` stops cleanly on `SIGINT` (Ctrl-C) or `SIGTERM`. Its workers
|
|
||||||
stop taking work, each finishing at most the directory listing or file
|
|
||||||
it is reading; the progress display is finished; and the records it has
|
|
||||||
hashed but not yet committed are committed, so the next scan does not
|
|
||||||
hash them again. Apart from that commit it starts no further writes or
|
|
||||||
deletions: records are deleted only after a complete walk, so those
|
|
||||||
under paths an interrupted walk never reached are kept. The database is
|
|
||||||
closed and the lock released as on any other exit, the line
|
|
||||||
`scan: interrupted after N files` goes to stderr, N being the number of
|
|
||||||
files the walk reached, and the exit code is 1. The next scan skips the
|
|
||||||
records already written and converges as usual.
|
|
||||||
|
|
||||||
After the first signal `scan` stops catching them, so a second one ends
|
|
||||||
it at once, as an uncaught signal does: the records not yet committed
|
|
||||||
are lost, and the database is left valid, as when any scan dies (see
|
|
||||||
"Database"). A `SIGINT` that `scan` inherits as ignored, as a script's
|
|
||||||
background job does, stays ignored.
|
|
||||||
|
|
||||||
## Entrypoints
|
## Entrypoints
|
||||||
|
|
||||||
This repository adheres to the
|
This repository adheres to the
|
||||||
|
|||||||
@@ -29,52 +29,6 @@
|
|||||||
|
|
||||||
# Completed Steps
|
# Completed Steps
|
||||||
|
|
||||||
- `.golangci.yml` replaced with the current canonical copy, which uses
|
|
||||||
`gomodguard_v2`, so lint no longer prints a deprecation warning
|
|
||||||
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/26)
|
|
||||||
|
|
||||||
- a test fails when either `hashWorker` cancellation check in `scan.go`
|
|
||||||
is removed (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/83)
|
|
||||||
|
|
||||||
- a database path holding `?`, `#` or `%` opens exactly the file it names
|
|
||||||
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/55)
|
|
||||||
|
|
||||||
- `scan` rejects `--workers` below 1 as a usage error instead of
|
|
||||||
running single-threaded (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/10)
|
|
||||||
|
|
||||||
- a test fails when either walk cancellation check in `scan.go` is
|
|
||||||
removed (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/81)
|
|
||||||
|
|
||||||
- test that `scan` refuses a database with another schema version
|
|
||||||
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/64)
|
|
||||||
|
|
||||||
- correct four inaccurate comments in `cancel_test.go` and rename
|
|
||||||
`walkCancelInFlightDirs` to `walkCancelInFlightFiles` (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/33)
|
|
||||||
|
|
||||||
- test the `-x` filesystem-boundary rules in `subdirJob` (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/17)
|
|
||||||
|
|
||||||
- `scan` creates the schema in one transaction; a version-0 database with a
|
|
||||||
`files` table is refused with a clear schema-version error (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/11)
|
|
||||||
|
|
||||||
- README documents install, Docker, a daily cron scan and how to read
|
|
||||||
and check the reports (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/54)
|
|
||||||
|
|
||||||
- the `Dockerfile` build stage keeps the Go module cache out of `builder`'s
|
|
||||||
home and copies the sources with `--chown`, so no `chown -R` walks them
|
|
||||||
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/43)
|
|
||||||
|
|
||||||
- `--version` prints `sfdupes VERSION` to stdout; README documents it and
|
|
||||||
`--help` (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/15)
|
|
||||||
|
|
||||||
- `scan` stops cleanly on `SIGINT` or `SIGTERM`: commits what it has
|
|
||||||
hashed, deletes nothing more, exits 1 (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/5)
|
|
||||||
|
|
||||||
- `report` and `trees` stream the records instead of holding them all in
|
- `report` and `trees` stream the records instead of holding them all in
|
||||||
memory; the schema gains the `files_signature` index (2026-10-04,
|
memory; the schema gains the `files_signature` index (2026-10-04,
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/14)
|
https://git.eeqj.de/sneak/sfdupes/issues/14)
|
||||||
|
|||||||
+49
-383
@@ -4,35 +4,20 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"database/sql"
|
"database/sql"
|
||||||
"errors"
|
"errors"
|
||||||
"fmt"
|
|
||||||
"os"
|
"os"
|
||||||
"os/signal"
|
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"slices"
|
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
|
||||||
"sync"
|
"sync"
|
||||||
"sync/atomic"
|
"sync/atomic"
|
||||||
"syscall"
|
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
"time"
|
||||||
)
|
)
|
||||||
|
|
||||||
// This file gathers the tests for scan cancellation and worker-pool
|
// poolUnwind bounds how long a goroutine is given to leave a pool
|
||||||
// unwinding. Everything it exercises lives in scan.go, so by the repo's
|
// after its context is cancelled. Only a failing run ever waits this
|
||||||
// convention of one test file per source file it would belong in
|
// long: a pool that ignored its cancellation parks forever, and this
|
||||||
// scan_test.go. It is kept separate on purpose: cancellation behaviour
|
// is what turns that into a failed assertion instead of a suite that
|
||||||
// cuts across both the walk pool and the hash pool as a single concern,
|
// hangs until the test binary's own timeout.
|
||||||
// and scan_test.go is already over 1,600 lines. That is the deliberate
|
|
||||||
// exception the convention otherwise expects to be stated.
|
|
||||||
|
|
||||||
// poolUnwind bounds how long a test waits for a cancellation to take
|
|
||||||
// effect: for a goroutine to return or a channel to close once its
|
|
||||||
// context is cancelled, or for a signal to cancel the scan's context.
|
|
||||||
// Only a failing run waits this long, and the bound is what makes that
|
|
||||||
// failure an assertion instead of a hang. A call made without it, as
|
|
||||||
// most of this file's scans are, has no bound: a regression that parks
|
|
||||||
// it is caught only as the test binary's own timeout.
|
|
||||||
const poolUnwind = 2 * time.Second
|
const poolUnwind = 2 * time.Second
|
||||||
|
|
||||||
// walkClock is a context whose cancellation is driven by the scan's
|
// walkClock is a context whose cancellation is driven by the scan's
|
||||||
@@ -43,14 +28,9 @@ const poolUnwind = 2 * time.Second
|
|||||||
//
|
//
|
||||||
// The accounting behind the n chosen by each test: every blocking
|
// The accounting behind the n chosen by each test: every blocking
|
||||||
// channel operation in the walk selects on Done, so the walk spends
|
// channel operation in the walk selects on Done, so the walk spends
|
||||||
// one consultation per file event plus a couple per directory. The
|
// one consultation per file event plus a couple per directory, while
|
||||||
// index load that runs ahead of it also consults Done, but a bounded
|
// the index load that runs ahead of it spends a small fixed number
|
||||||
// number of times that does not grow with the record count. The tests
|
// (three) whatever the record count.
|
||||||
// depend on that property, not on the bound's exact value: each test
|
|
||||||
// sets n from the consultations of the walk, plus those of the hash
|
|
||||||
// phase when it cancels mid-hash, far from both ends of the phase it
|
|
||||||
// interrupts, so the cancellation lands inside that phase whatever the
|
|
||||||
// record count.
|
|
||||||
type walkClock struct {
|
type walkClock struct {
|
||||||
n int64
|
n int64
|
||||||
seen atomic.Int64
|
seen atomic.Int64
|
||||||
@@ -106,14 +86,11 @@ func (c *walkClock) Value(_ any) any {
|
|||||||
// directory still queued and only the handful already in flight can
|
// directory still queued and only the handful already in flight can
|
||||||
// emit anything more.
|
// emit anything more.
|
||||||
const (
|
const (
|
||||||
walkCancelDirs = 100
|
walkCancelDirs = 100
|
||||||
walkCancelFilesPerDir = 20
|
walkCancelFilesPerDir = 20
|
||||||
walkCancelFiles = walkCancelDirs * walkCancelFilesPerDir
|
walkCancelFiles = walkCancelDirs * walkCancelFilesPerDir
|
||||||
walkCancelWorkers = 4
|
walkCancelWorkers = 4
|
||||||
// The most files the walkCancelWorkers directories already in
|
walkCancelInFlightDirs = walkCancelWorkers * walkCancelFilesPerDir
|
||||||
// flight when the scan is cancelled can still emit, at
|
|
||||||
// walkCancelFilesPerDir each. A file count, not a directory count.
|
|
||||||
walkCancelInFlightFiles = walkCancelWorkers * walkCancelFilesPerDir
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// walkCancelAtDone is the consultation on which the fixture's context
|
// walkCancelAtDone is the consultation on which the fixture's context
|
||||||
@@ -173,13 +150,9 @@ func assertRecordsIntact(t *testing.T, db *sql.DB, before []string) {
|
|||||||
// Every one of those records would look vanished to the update phase.
|
// Every one of those records would look vanished to the update phase.
|
||||||
// The guard is what stops the scan there, and this test is what
|
// The guard is what stops the scan there, and this test is what
|
||||||
// notices if it stops doing so: deleting the guard, or making it
|
// notices if it stops doing so: deleting the guard, or making it
|
||||||
// unreachable, makes the scan carry its truncated view into the update
|
// unreachable, makes the scan carry its truncated view into a later
|
||||||
// phase, which counts every record the walk never reached for removal.
|
// phase and fail there instead, with a wrapped error rather than the
|
||||||
//
|
// bare cancellation.
|
||||||
// The syncScan call here is not bounded by poolUnwind: a regression
|
|
||||||
// that left a worker pool parked would hang it, and that regression is
|
|
||||||
// caught only by the test binary's own timeout, not by a quick
|
|
||||||
// assertion.
|
|
||||||
//
|
//
|
||||||
//nolint:paralleltest // counts goroutines: must not run beside others
|
//nolint:paralleltest // counts goroutines: must not run beside others
|
||||||
func TestSyncScanCancelledMidWalkKeepsRecords(t *testing.T) {
|
func TestSyncScanCancelledMidWalkKeepsRecords(t *testing.T) {
|
||||||
@@ -209,10 +182,10 @@ func TestSyncScanCancelledMidWalkKeepsRecords(t *testing.T) {
|
|||||||
|
|
||||||
// assertWalkGuardAborted checks that the scan stopped at the post-walk
|
// assertWalkGuardAborted checks that the scan stopped at the post-walk
|
||||||
// guard: with a census that is neither empty (the walk really ran)
|
// guard: with a census that is neither empty (the walk really ran)
|
||||||
// nor complete (it really was cut short), and with no record counted
|
// nor complete (it really was cut short), and with the guard's own
|
||||||
// for removal. A removal count means the partial census was carried
|
// bare cancellation as the error. A wrapped error means the partial
|
||||||
// past the guard into the update phase, which is the failure this test
|
// census was carried past the guard into the hash or update phase,
|
||||||
// exists to catch.
|
// which is the failure this test exists to catch.
|
||||||
func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
|
func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
@@ -221,6 +194,12 @@ func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
|
|||||||
err, context.Canceled)
|
err, context.Canceled)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if errors.Unwrap(err) != nil {
|
||||||
|
t.Errorf("syncScan reported %q, want the guard's bare "+
|
||||||
|
"cancellation: a wrapped error means the truncated census "+
|
||||||
|
"reached a later phase", err)
|
||||||
|
}
|
||||||
|
|
||||||
if st.unchanged == 0 {
|
if st.unchanged == 0 {
|
||||||
t.Fatalf("stats = %+v: the census is empty, so the walk never "+
|
t.Fatalf("stats = %+v: the census is empty, so the walk never "+
|
||||||
"ran and the guard was reached for the wrong reason", st)
|
"ran and the guard was reached for the wrong reason", st)
|
||||||
@@ -232,11 +211,10 @@ func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// The workers drop every directory still queued once the scan is
|
// The workers drop every directory still queued once the scan is
|
||||||
// cancelled, so only the files in the directories already in flight
|
// cancelled, so only the directories already in flight can add to
|
||||||
// can add to the census after the fact. A census beyond that bound
|
// the census after the fact. A census beyond that bound would mean
|
||||||
// would mean the cancellation was not observed where it should have
|
// the cancellation was not observed where it should have been.
|
||||||
// been.
|
limit := walkCancelAtDone + walkCancelInFlightDirs
|
||||||
limit := walkCancelAtDone + walkCancelInFlightFiles
|
|
||||||
if st.unchanged > limit {
|
if st.unchanged > limit {
|
||||||
t.Errorf("census covers %d files, want at most %d: the walk kept "+
|
t.Errorf("census covers %d files, want at most %d: the walk kept "+
|
||||||
"taking directories off the queue after cancellation",
|
"taking directories off the queue after cancellation",
|
||||||
@@ -282,242 +260,6 @@ func TestSyncScanCancelledBeforeLoadIndex(t *testing.T) {
|
|||||||
assertRecordsIntact(t, db, before)
|
assertRecordsIntact(t, db, before)
|
||||||
}
|
}
|
||||||
|
|
||||||
// hashCancelAtDone is the consultation on which the mid-hash test's
|
|
||||||
// context cancels itself. The walk of buildWalkCancelTree spends about
|
|
||||||
// one per file and three per directory, and the hash phase then one per
|
|
||||||
// file hashed, so this lands about half way through the hash phase.
|
|
||||||
const hashCancelAtDone = walkCancelFiles + 3*walkCancelDirs +
|
|
||||||
walkCancelFiles/2
|
|
||||||
|
|
||||||
// TestSyncScanCancelledMidHashKeepsHashedRecords cancels a first scan
|
|
||||||
// part-way through its hash phase. The fixture holds fewer files than a
|
|
||||||
// batch, so every file hashed is still waiting to be committed: the scan
|
|
||||||
// must commit them all before it returns, and the next scan must hash
|
|
||||||
// only the rest.
|
|
||||||
func TestSyncScanCancelledMidHashKeepsHashedRecords(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
dir := buildWalkCancelTree(t)
|
|
||||||
db := openTestDB(t)
|
|
||||||
|
|
||||||
st, err := syncScan(newWalkClock(hashCancelAtDone), db,
|
|
||||||
[]string{dir}, walkCancelWorkers, false)
|
|
||||||
if !errors.Is(err, context.Canceled) {
|
|
||||||
t.Fatalf("syncScan cancelled mid-hash = %v, want %v",
|
|
||||||
err, context.Canceled)
|
|
||||||
}
|
|
||||||
|
|
||||||
if st.walked != walkCancelFiles || st.added == 0 ||
|
|
||||||
st.added >= walkCancelFiles {
|
|
||||||
t.Fatalf("stats = %+v: want the walk complete and the hash phase "+
|
|
||||||
"cut short", st)
|
|
||||||
}
|
|
||||||
|
|
||||||
if got := len(dbRecords(t, db)); got != st.added {
|
|
||||||
t.Errorf("%d records after the cancelled scan, want the %d it hashed",
|
|
||||||
got, st.added)
|
|
||||||
}
|
|
||||||
|
|
||||||
hashed := st.added
|
|
||||||
|
|
||||||
st = syncTree(t, db, dir)
|
|
||||||
if st.added != walkCancelFiles-hashed || st.unchanged != hashed {
|
|
||||||
t.Errorf("next scan stats = %+v, want %d added %d unchanged",
|
|
||||||
st, walkCancelFiles-hashed, hashed)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// storedPaths opens the database at path as report does, which fails
|
|
||||||
// unless it is a valid database, and returns its records' paths.
|
|
||||||
func storedPaths(t *testing.T, path string) []string {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
db, err := openReportDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = db.Close() }()
|
|
||||||
|
|
||||||
return recordPaths(dbRecords(t, db))
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestRunScanInterrupted calls the scan entrypoint with a context that
|
|
||||||
// is already cancelled, as when a signal arrives at once. It must return
|
|
||||||
// errInterrupted promptly with its one line on stderr, leave the
|
|
||||||
// database valid and as it was, and leave nothing in the way of the
|
|
||||||
// next scan, which must bring the database up to date.
|
|
||||||
func TestRunScanInterrupted(t *testing.T) {
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
stderr := captureStderr(t)
|
|
||||||
dir := buildSmokeTree(t)
|
|
||||||
|
|
||||||
err := runScan(t.Context(), []string{dir}, walkCancelWorkers, false)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
before := storedPaths(t, path)
|
|
||||||
|
|
||||||
// A vanished file and a new one: the interrupted scan records
|
|
||||||
// neither.
|
|
||||||
gone := filepath.Join(dir, "a", "unique.bin")
|
|
||||||
|
|
||||||
err = os.Remove(gone)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
added := writeFile(t, dir, "a/new.bin", pattern(50, 10))
|
|
||||||
shown := len(stderr())
|
|
||||||
done := make(chan struct{})
|
|
||||||
|
|
||||||
go func() {
|
|
||||||
defer close(done)
|
|
||||||
|
|
||||||
err = runScan(cancelledContext(t), []string{dir}, walkCancelWorkers,
|
|
||||||
false)
|
|
||||||
}()
|
|
||||||
|
|
||||||
awaitReturn(t, done, "runScan")
|
|
||||||
|
|
||||||
if !errors.Is(err, errInterrupted) {
|
|
||||||
t.Fatalf("runScan on a cancelled context = %v, want %v",
|
|
||||||
err, errInterrupted)
|
|
||||||
}
|
|
||||||
|
|
||||||
want := "scan: interrupted after 0 files\n"
|
|
||||||
if got := stderr()[shown:]; got != want {
|
|
||||||
t.Errorf("stderr = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
assertNoSidecars(t, path)
|
|
||||||
|
|
||||||
if got := storedPaths(t, path); !slices.Equal(got, before) {
|
|
||||||
t.Errorf("records = %q after the interrupted scan, want %q",
|
|
||||||
got, before)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = runScan(t.Context(), []string{dir}, walkCancelWorkers, false)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
got := storedPaths(t, path)
|
|
||||||
if slices.Contains(got, gone) || !slices.Contains(got, added) {
|
|
||||||
t.Errorf("records = %q after the next scan, want %q gone and %q "+
|
|
||||||
"added", got, gone, added)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestRunScanInterruptedMidHash interrupts the scan entrypoint part-way
|
|
||||||
// through its hash phase, after the database is open. It must return
|
|
||||||
// errInterrupted, release the lock, end stderr with its line counting
|
|
||||||
// every file the walk reached, close the database out of WAL mode, and
|
|
||||||
// keep the records it hashed.
|
|
||||||
func TestRunScanInterruptedMidHash(t *testing.T) {
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
stderr := captureStderr(t)
|
|
||||||
dir := buildWalkCancelTree(t)
|
|
||||||
|
|
||||||
err := runScan(newWalkClock(hashCancelAtDone), []string{dir},
|
|
||||||
walkCancelWorkers, false)
|
|
||||||
if !errors.Is(err, errInterrupted) {
|
|
||||||
t.Fatalf("runScan interrupted mid-hash = %v, want %v",
|
|
||||||
err, errInterrupted)
|
|
||||||
}
|
|
||||||
|
|
||||||
holdScanLock(t, path)
|
|
||||||
|
|
||||||
want := fmt.Sprintf("scan: interrupted after %d files\n", walkCancelFiles)
|
|
||||||
if got := stderr(); !strings.HasSuffix(got, want) {
|
|
||||||
t.Errorf("stderr = %q, want it to end with %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
assertNoSidecars(t, path)
|
|
||||||
|
|
||||||
db, err := openReportDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = db.Close() }()
|
|
||||||
|
|
||||||
// A plain close also removes the sidecars, but leaves WAL mode on.
|
|
||||||
var mode string
|
|
||||||
|
|
||||||
err = db.QueryRowContext(t.Context(), "PRAGMA journal_mode").Scan(&mode)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
if mode != "delete" {
|
|
||||||
t.Errorf("journal mode = %q after the interrupted scan, want %q",
|
|
||||||
mode, "delete")
|
|
||||||
}
|
|
||||||
|
|
||||||
kept := len(dbRecords(t, db))
|
|
||||||
if kept == 0 || kept >= walkCancelFiles {
|
|
||||||
t.Errorf("%d records after the interrupted scan, want those it "+
|
|
||||||
"hashed: some but not all of the %d files", kept, walkCancelFiles)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestInterruptContextCatchesSIGTERM sends SIGTERM to the test process
|
|
||||||
// while the scan's handler is installed, and checks that it cancels the
|
|
||||||
// scan's context.
|
|
||||||
//
|
|
||||||
//nolint:paralleltest // signals the whole process: must not run beside a scan
|
|
||||||
func TestInterruptContextCatchesSIGTERM(t *testing.T) {
|
|
||||||
// Caught here as well, so that a handler that misses SIGTERM fails
|
|
||||||
// this test instead of ending the test process.
|
|
||||||
caught := make(chan os.Signal, 1)
|
|
||||||
signal.Notify(caught, syscall.SIGTERM)
|
|
||||||
|
|
||||||
defer signal.Stop(caught)
|
|
||||||
|
|
||||||
ctx, stop := interruptContext(t.Context())
|
|
||||||
defer stop()
|
|
||||||
|
|
||||||
err := syscall.Kill(os.Getpid(), syscall.SIGTERM)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
select {
|
|
||||||
case <-ctx.Done():
|
|
||||||
case <-time.After(poolUnwind):
|
|
||||||
t.Fatal("SIGTERM did not cancel the scan's context")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestCommitFullBatchKeepsFailedBatch checks that a full batch whose
|
|
||||||
// commit fails, as it does once the scan is interrupted, stays in the
|
|
||||||
// batch, so that syncScan's final commit saves it.
|
|
||||||
func TestCommitFullBatchKeepsFailedBatch(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
s := &scanState{db: openTestDB(t)}
|
|
||||||
for i := range updateBatchSize {
|
|
||||||
s.batch = append(s.batch, scanRec{path: "/f" + strconv.Itoa(i)})
|
|
||||||
}
|
|
||||||
|
|
||||||
err := s.commitFullBatch(cancelledContext(t))
|
|
||||||
if !errors.Is(err, context.Canceled) {
|
|
||||||
t.Fatalf("commitFullBatch on a cancelled context = %v, want %v",
|
|
||||||
err, context.Canceled)
|
|
||||||
}
|
|
||||||
|
|
||||||
if len(s.batch) != updateBatchSize {
|
|
||||||
t.Errorf("batch holds %d records after the failed commit, want %d",
|
|
||||||
len(s.batch), updateBatchSize)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// drainClosed counts the values received from ch until it closes,
|
// drainClosed counts the values received from ch until it closes,
|
||||||
// failing the test if it does not close within poolUnwind. A pool that
|
// failing the test if it does not close within poolUnwind. A pool that
|
||||||
// ignored its cancellation leaves its channel open with its goroutines
|
// ignored its cancellation leaves its channel open with its goroutines
|
||||||
@@ -586,49 +328,20 @@ func TestSendEventAbandonsBlockedSend(t *testing.T) {
|
|||||||
awaitReturn(t, done, "sendEvent")
|
awaitReturn(t, done, "sendEvent")
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestWalkOneDirStopsWhenCancelled checks that a cancelled scan stops
|
|
||||||
// reading a directory instead of going through the rest of its
|
|
||||||
// entries. A walk that kept going would return the subdirectory below
|
|
||||||
// to descend into. Unlike a file event, that return is not a send the
|
|
||||||
// cancellation can abandon, so the test catches the regression every
|
|
||||||
// time.
|
|
||||||
func TestWalkOneDirStopsWhenCancelled(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
dir := t.TempDir()
|
|
||||||
|
|
||||||
err := os.Mkdir(filepath.Join(dir, "sub"), 0o750)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Unbuffered and unread: on a cancelled scan every send gives up.
|
|
||||||
events := make(chan walkEvent)
|
|
||||||
|
|
||||||
subs := walkOneDir(cancelledContext(t), dirJob{path: dir}, false, events)
|
|
||||||
if len(subs) != 0 {
|
|
||||||
t.Errorf("cancelled walkOneDir returned %+v to descend into, "+
|
|
||||||
"want none", subs)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestWalkWorkersDropQueuedDirs checks that cancelled walk workers keep
|
// TestWalkWorkersDropQueuedDirs checks that cancelled walk workers keep
|
||||||
// reading jobs and drop the directories rather than stopping their
|
// reading jobs and drop the directories rather than stopping their
|
||||||
// read: the range over jobs has to run out for the pool to tear down
|
// read: the range over jobs has to run out for the pool to tear down
|
||||||
// and close its event stream. The queued directory does not exist, so
|
// and close its event stream.
|
||||||
// a worker that walked it anyway would send a warning before
|
|
||||||
// walkOneDir's own cancellation check could stop it. On a cancelled
|
|
||||||
// scan that send delivers or gives up at random, so with 64 jobs
|
|
||||||
// queued the regression has a one in 2^64 chance of passing.
|
|
||||||
func TestWalkWorkersDropQueuedDirs(t *testing.T) {
|
func TestWalkWorkersDropQueuedDirs(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
missing := filepath.Join(t.TempDir(), "missing")
|
dir := t.TempDir()
|
||||||
|
writeEmptyFiles(t, dir, walkCancelFilesPerDir)
|
||||||
|
|
||||||
jobs, _, events := startWalkWorkers(cancelledContext(t), 2, false)
|
jobs, _, events := startWalkWorkers(cancelledContext(t), 2, false)
|
||||||
|
|
||||||
for range 64 {
|
for range 4 {
|
||||||
jobs <- dirJob{path: missing}
|
jobs <- dirJob{path: dir}
|
||||||
}
|
}
|
||||||
|
|
||||||
close(jobs)
|
close(jobs)
|
||||||
@@ -699,11 +412,7 @@ func TestDispatchDirsClosesJobsWhenCancelled(t *testing.T) {
|
|||||||
|
|
||||||
// TestFeedHashJobsClosesJobsWhenCancelled checks that the hash feeder
|
// TestFeedHashJobsClosesJobsWhenCancelled checks that the hash feeder
|
||||||
// abandons the runs it has not queued yet and still closes the job
|
// abandons the runs it has not queued yet and still closes the job
|
||||||
// channel, which is what lets the workers' range terminate. The
|
// channel, which is what lets the workers' range terminate.
|
||||||
// receive on jobs below is not bounded: a feeder that returned without
|
|
||||||
// closing jobs would leave that receive with no sender and no close, so
|
|
||||||
// this regression is caught by the test binary's timeout rather than by
|
|
||||||
// a bounded assertion.
|
|
||||||
func TestFeedHashJobsClosesJobsWhenCancelled(t *testing.T) {
|
func TestFeedHashJobsClosesJobsWhenCancelled(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
@@ -729,84 +438,41 @@ func TestFeedHashJobsClosesJobsWhenCancelled(t *testing.T) {
|
|||||||
// TestHashWorkerDropsQueuedRuns checks that a cancelled hash worker
|
// TestHashWorkerDropsQueuedRuns checks that a cancelled hash worker
|
||||||
// keeps reading jobs and drops the runs rather than reading files
|
// keeps reading jobs and drops the runs rather than reading files
|
||||||
// nobody wants the hashes of — while still letting the range run out
|
// nobody wants the hashes of — while still letting the range run out
|
||||||
// so the pool tears down. The hash function records that it was
|
// so the pool tears down. The queued run names a file that does not
|
||||||
// called, so a worker that hashed the queued run anyway is caught
|
// exist, so a worker that hashed it anyway would produce a result.
|
||||||
// every time.
|
|
||||||
func TestHashWorkerDropsQueuedRuns(t *testing.T) {
|
func TestHashWorkerDropsQueuedRuns(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
done := make(chan struct{})
|
done := make(chan struct{})
|
||||||
jobs := make(chan []fileRec, 1)
|
jobs := make(chan []fileRec, 1)
|
||||||
results := make(chan hashResult)
|
results := make(chan hashResult, 1)
|
||||||
|
|
||||||
jobs <- []fileRec{{path: filepath.Join(t.TempDir(), "missing"), size: 1}}
|
run := []fileRec{{path: filepath.Join(t.TempDir(), "missing"), size: 1}}
|
||||||
|
|
||||||
|
jobs <- run
|
||||||
|
|
||||||
close(jobs)
|
close(jobs)
|
||||||
|
|
||||||
var hashed atomic.Bool
|
|
||||||
|
|
||||||
hash := func(path string, size int64) (string, string, string, error) {
|
|
||||||
hashed.Store(true)
|
|
||||||
|
|
||||||
return hashSignature(path, size)
|
|
||||||
}
|
|
||||||
|
|
||||||
go func() {
|
go func() {
|
||||||
defer close(done)
|
defer close(done)
|
||||||
|
|
||||||
hashWorker(cancelledContext(t), jobs, results, hash)
|
hashWorker(cancelledContext(t), jobs, results, hashSignature)
|
||||||
}()
|
}()
|
||||||
|
|
||||||
awaitReturn(t, done, "hashWorker")
|
awaitReturn(t, done, "hashWorker")
|
||||||
|
|
||||||
if hashed.Load() {
|
select {
|
||||||
t.Error("cancelled hash worker hashed the queued run, want it dropped")
|
case r := <-results:
|
||||||
|
t.Errorf("cancelled hash worker produced %+v, want the run dropped",
|
||||||
|
r)
|
||||||
|
default:
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestHashWorkerAbandonsBlockedSend checks that a hash worker with a
|
|
||||||
// result to deliver and nobody to deliver it to leaves once the scan
|
|
||||||
// is cancelled, instead of holding the pool open. The scan tests do
|
|
||||||
// not catch this: stop drains results, which frees a parked worker
|
|
||||||
// anyway.
|
|
||||||
func TestHashWorkerAbandonsBlockedSend(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
ctx, cancel := context.WithCancel(t.Context())
|
|
||||||
defer cancel()
|
|
||||||
|
|
||||||
done := make(chan struct{})
|
|
||||||
jobs := make(chan []fileRec, 1)
|
|
||||||
// Unbuffered and unread, with jobs left open: the worker's only way
|
|
||||||
// out is the cancellation case beside its send.
|
|
||||||
results := make(chan hashResult)
|
|
||||||
|
|
||||||
jobs <- []fileRec{{path: filepath.Join(t.TempDir(), "missing"), size: 1}}
|
|
||||||
|
|
||||||
// The scan is cancelled while the worker hashes, so the worker has
|
|
||||||
// already passed the check that drops queued runs.
|
|
||||||
hash := func(path string, size int64) (string, string, string, error) {
|
|
||||||
cancel()
|
|
||||||
|
|
||||||
return hashSignature(path, size)
|
|
||||||
}
|
|
||||||
|
|
||||||
go func() {
|
|
||||||
defer close(done)
|
|
||||||
|
|
||||||
hashWorker(ctx, jobs, results, hash)
|
|
||||||
}()
|
|
||||||
|
|
||||||
awaitReturn(t, done, "hashWorker")
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestHashPhaseCancelledReturnsContextError checks the result loop's
|
// TestHashPhaseCancelledReturnsContextError checks the result loop's
|
||||||
// own exit: with the pool cancelled, no result will ever arrive, and
|
// own exit: with the pool cancelled, no result will ever arrive, and
|
||||||
// the loop must leave through the cancellation rather than wait for a
|
// the loop must leave through the cancellation rather than wait for a
|
||||||
// receive that cannot happen. This call is not bounded by poolUnwind: a
|
// receive that cannot happen.
|
||||||
// loop that dropped its cancellation case would block on that receive,
|
|
||||||
// so the regression surfaces as the test binary's timeout rather than
|
|
||||||
// as a bounded assertion.
|
|
||||||
func TestHashPhaseCancelledReturnsContextError(t *testing.T) {
|
func TestHashPhaseCancelledReturnsContextError(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
|
|||||||
@@ -6,7 +6,6 @@ import (
|
|||||||
"errors"
|
"errors"
|
||||||
"fmt"
|
"fmt"
|
||||||
"io/fs"
|
"io/fs"
|
||||||
"net/url"
|
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"slices"
|
"slices"
|
||||||
@@ -108,19 +107,7 @@ const reportParams = "mode=ro" +
|
|||||||
// openDB opens the SQLite database at path with the connection
|
// openDB opens the SQLite database at path with the connection
|
||||||
// parameters params. It does not create or verify the schema.
|
// parameters params. It does not create or verify the schema.
|
||||||
func openDB(path, params string) (*sql.DB, error) {
|
func openDB(path, params string) (*sql.DB, error) {
|
||||||
// The path is escaped into a file: URI, so ?, # and % in it stay
|
db, err := sql.Open("sqlite", "file:"+path+"?"+params)
|
||||||
// part of the file name. SQLite reads what follows file:// up to
|
|
||||||
// the next / as a host name, so an absolute path goes after an
|
|
||||||
// empty host (file:///abs) and a relative path goes without one
|
|
||||||
// (file:rel).
|
|
||||||
uri := url.URL{
|
|
||||||
Scheme: "file",
|
|
||||||
OmitHost: !filepath.IsAbs(path),
|
|
||||||
Path: path,
|
|
||||||
RawQuery: params,
|
|
||||||
}
|
|
||||||
|
|
||||||
db, err := sql.Open("sqlite", uri.String())
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("open database %s: %w", path, err)
|
return nil, fmt.Errorf("open database %s: %w", path, err)
|
||||||
}
|
}
|
||||||
@@ -232,12 +219,6 @@ func openReportDatabase(ctx context.Context,
|
|||||||
}
|
}
|
||||||
|
|
||||||
v, err := userVersion(ctx, db)
|
v, err := userVersion(ctx, db)
|
||||||
if err == nil && v == 0 {
|
|
||||||
// An empty database passes this check and fails the version
|
|
||||||
// check below.
|
|
||||||
err = checkUnversioned(ctx, db)
|
|
||||||
}
|
|
||||||
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
_ = db.Close()
|
_ = db.Close()
|
||||||
|
|
||||||
@@ -264,11 +245,6 @@ func initSchema(ctx context.Context, db *sql.DB) error {
|
|||||||
|
|
||||||
switch v {
|
switch v {
|
||||||
case 0:
|
case 0:
|
||||||
err = checkUnversioned(ctx, db)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
|
|
||||||
return createSchema(ctx, db)
|
return createSchema(ctx, db)
|
||||||
case schemaVersion:
|
case schemaVersion:
|
||||||
return nil
|
return nil
|
||||||
@@ -278,65 +254,25 @@ func initSchema(ctx context.Context, db *sql.DB) error {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// checkUnversioned checks a database at user_version 0 before it is
|
|
||||||
// taken for an empty one. createSchema creates the files table and
|
|
||||||
// sets the version together, so a files table at version 0 was made by
|
|
||||||
// something else. Adopting it could corrupt unrelated data, so that is
|
|
||||||
// a schema-version error telling the operator to remove the file and
|
|
||||||
// rescan.
|
|
||||||
func checkUnversioned(ctx context.Context, db *sql.DB) error {
|
|
||||||
var name string
|
|
||||||
|
|
||||||
err := db.QueryRowContext(ctx,
|
|
||||||
"SELECT name FROM sqlite_master "+
|
|
||||||
"WHERE type = 'table' AND name = 'files'").Scan(&name)
|
|
||||||
|
|
||||||
switch {
|
|
||||||
case err == nil:
|
|
||||||
return fmt.Errorf(
|
|
||||||
"has a files table but no schema version; "+
|
|
||||||
"remove the file and rescan: %w", errSchemaVersion)
|
|
||||||
case errors.Is(err, sql.ErrNoRows):
|
|
||||||
return nil
|
|
||||||
default:
|
|
||||||
return fmt.Errorf("check for files table: %w", err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// createSchema applies the schema to a fresh database and stamps the
|
// createSchema applies the schema to a fresh database and stamps the
|
||||||
// schema version in one transaction, so a creation stopped partway, by
|
// schema version.
|
||||||
// an interrupt or an error, leaves an empty database the next scan
|
|
||||||
// sets up, never a files table at version 0, which checkUnversioned
|
|
||||||
// refuses.
|
|
||||||
func createSchema(ctx context.Context, db *sql.DB) error {
|
func createSchema(ctx context.Context, db *sql.DB) error {
|
||||||
tx, err := db.BeginTx(ctx, nil)
|
_, err := db.ExecContext(ctx, createTableSQL)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("create schema: %w", err)
|
return fmt.Errorf("create schema: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
defer func() { _ = tx.Rollback() }()
|
_, err = db.ExecContext(ctx, createIndexSQL)
|
||||||
|
|
||||||
_, err = tx.ExecContext(ctx, createTableSQL)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("create schema: %w", err)
|
return fmt.Errorf("create schema: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
_, err = tx.ExecContext(ctx, createIndexSQL)
|
_, err = db.ExecContext(ctx,
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("create schema: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err = tx.ExecContext(ctx,
|
|
||||||
"PRAGMA user_version = "+strconv.Itoa(schemaVersion))
|
"PRAGMA user_version = "+strconv.Itoa(schemaVersion))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("set schema version: %w", err)
|
return fmt.Errorf("set schema version: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
err = tx.Commit()
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("create schema: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+2
-80
@@ -8,7 +8,6 @@ import (
|
|||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"slices"
|
"slices"
|
||||||
"strings"
|
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -78,76 +77,6 @@ func TestOpenScanDatabaseCreates(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestOpenDatabaseUnversionedForeign(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
// A database that has a files table but user_version 0, written by
|
|
||||||
// some other tool. report, trees and scan must refuse it with the
|
|
||||||
// schema-version error, not adopt it and not emit a raw SQLite
|
|
||||||
// "table files already exists".
|
|
||||||
path := testDBPath(t)
|
|
||||||
|
|
||||||
db, err := sql.Open("sqlite", path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err = db.ExecContext(t.Context(), "CREATE TABLE files (x INTEGER)")
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
_ = db.Close()
|
|
||||||
|
|
||||||
_, err = openReportDatabase(t.Context(), path)
|
|
||||||
if !errors.Is(err, errSchemaVersion) ||
|
|
||||||
!strings.Contains(err.Error(), "remove the file and rescan") {
|
|
||||||
t.Fatalf("report: err = %v, want errSchemaVersion telling the "+
|
|
||||||
"operator to remove the file and rescan", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err = openScanDatabase(t.Context(), path)
|
|
||||||
if !errors.Is(err, errSchemaVersion) ||
|
|
||||||
!strings.Contains(err.Error(), "remove the file and rescan") {
|
|
||||||
t.Fatalf("scan: err = %v, want errSchemaVersion telling the "+
|
|
||||||
"operator to remove the file and rescan", err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestSchemaCreationStoppedPartway(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
// A first scan stopped while creating the schema must leave a
|
|
||||||
// database the next scan accepts. max_page_count(2) leaves room for
|
|
||||||
// the files table but not its index, so schema creation fails right
|
|
||||||
// after CREATE TABLE, a point an interrupt could also stop it at.
|
|
||||||
path := testDBPath(t)
|
|
||||||
|
|
||||||
db, err := openDB(path, scanParams+"&_pragma=max_page_count(2)")
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = initSchema(t.Context(), db)
|
|
||||||
_ = db.Close()
|
|
||||||
|
|
||||||
if err == nil {
|
|
||||||
t.Fatal("initSchema with no room for the index succeeded")
|
|
||||||
}
|
|
||||||
|
|
||||||
db, err = openScanDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("next scan: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = db.Close() }()
|
|
||||||
|
|
||||||
v, err := userVersion(t.Context(), db)
|
|
||||||
if err != nil || v != schemaVersion {
|
|
||||||
t.Fatalf("userVersion = %d, %v; want %d, nil", v, err, schemaVersion)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestOpenReportDatabaseMissing(t *testing.T) {
|
func TestOpenReportDatabaseMissing(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
@@ -157,11 +86,9 @@ func TestOpenReportDatabaseMissing(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestOpenDatabaseVersionMismatch(t *testing.T) {
|
func TestOpenReportDatabaseVersionMismatch(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// A database stamped with a schema version other than 0 and
|
|
||||||
// schemaVersion. report, trees and scan must all refuse it.
|
|
||||||
path := testDBPath(t)
|
path := testDBPath(t)
|
||||||
|
|
||||||
db, err := openScanDatabase(t.Context(), path)
|
db, err := openScanDatabase(t.Context(), path)
|
||||||
@@ -178,12 +105,7 @@ func TestOpenDatabaseVersionMismatch(t *testing.T) {
|
|||||||
|
|
||||||
_, err = openReportDatabase(t.Context(), path)
|
_, err = openReportDatabase(t.Context(), path)
|
||||||
if !errors.Is(err, errSchemaVersion) {
|
if !errors.Is(err, errSchemaVersion) {
|
||||||
t.Fatalf("report: err = %v, want errSchemaVersion", err)
|
t.Fatalf("err = %v, want errSchemaVersion", err)
|
||||||
}
|
|
||||||
|
|
||||||
_, err = openScanDatabase(t.Context(), path)
|
|
||||||
if !errors.Is(err, errSchemaVersion) {
|
|
||||||
t.Fatalf("scan: err = %v, want errSchemaVersion", err)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -15,7 +15,6 @@
|
|||||||
// sfdupes scan [--workers N] [-x] PATH...
|
// sfdupes scan [--workers N] [-x] PATH...
|
||||||
// sfdupes report > dupes.tsv
|
// sfdupes report > dupes.tsv
|
||||||
// sfdupes trees > dupetrees.tsv
|
// sfdupes trees > dupetrees.tsv
|
||||||
// sfdupes --version
|
|
||||||
//
|
//
|
||||||
// See README.md for the complete specification.
|
// See README.md for the complete specification.
|
||||||
package main
|
package main
|
||||||
@@ -51,10 +50,6 @@ const (
|
|||||||
// cobra prints for it is the whole message.
|
// cobra prints for it is the whole message.
|
||||||
var errNoSubcommand = errors.New("no subcommand")
|
var errNoSubcommand = errors.New("no subcommand")
|
||||||
|
|
||||||
// errWorkersBelowOne is the usage error for a scan --workers value
|
|
||||||
// below 1.
|
|
||||||
var errWorkersBelowOne = errors.New("--workers must be at least 1")
|
|
||||||
|
|
||||||
// Version is the build version, injected at link time via -ldflags
|
// Version is the build version, injected at link time via -ldflags
|
||||||
// (see the Makefile); "dev" for a plain go build.
|
// (see the Makefile); "dev" for a plain go build.
|
||||||
//
|
//
|
||||||
@@ -92,9 +87,6 @@ func run(args []string, stdout, stderr io.Writer) int {
|
|||||||
switch {
|
switch {
|
||||||
case err == nil:
|
case err == nil:
|
||||||
return exitOK
|
return exitOK
|
||||||
case errors.Is(err, errInterrupted):
|
|
||||||
// The interrupted scan has printed its own line.
|
|
||||||
return exitFatal
|
|
||||||
case errors.As(err, &fatal):
|
case errors.As(err, &fatal):
|
||||||
// The command ran and failed: a runtime error, reported
|
// The command ran and failed: a runtime error, reported
|
||||||
// without the usage text that a usage error gets.
|
// without the usage text that a usage error gets.
|
||||||
@@ -102,35 +94,22 @@ func run(args []string, stdout, stderr io.Writer) int {
|
|||||||
|
|
||||||
return exitFatal
|
return exitFatal
|
||||||
default:
|
default:
|
||||||
// A usage error, which cobra has already reported on stderr.
|
// A usage error: cobra has already printed the message and
|
||||||
|
// the usage text.
|
||||||
return exitUsage
|
return exitUsage
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// newRootCommand builds the command tree. Everything on stdout is
|
// newRootCommand builds the command tree. Everything on stdout is
|
||||||
// machine-readable data, the version line included; all human-facing
|
// machine-readable data; all human-facing output (help, usage, errors)
|
||||||
// output (help, usage, errors) goes to stderr.
|
// goes to stderr.
|
||||||
func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
||||||
var showVersion bool
|
|
||||||
|
|
||||||
printVersion := runE(func(context.Context, []string) error {
|
|
||||||
_, err := fmt.Fprintf(stdout, "sfdupes %s\n", Version)
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("write stdout: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return nil
|
|
||||||
})
|
|
||||||
|
|
||||||
root := &cobra.Command{
|
root := &cobra.Command{
|
||||||
Use: "sfdupes",
|
Use: "sfdupes",
|
||||||
Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
|
Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
|
||||||
Args: cobra.NoArgs,
|
Version: Version,
|
||||||
RunE: func(cmd *cobra.Command, args []string) error {
|
Args: cobra.NoArgs,
|
||||||
if showVersion {
|
RunE: func(cmd *cobra.Command, _ []string) error {
|
||||||
return printVersion(cmd, args)
|
|
||||||
}
|
|
||||||
|
|
||||||
// A missing subcommand prints usage and exits 2: cobra
|
// A missing subcommand prints usage and exits 2: cobra
|
||||||
// prints the usage text for the returned error, and run
|
// prints the usage text for the returned error, and run
|
||||||
// maps everything that is not a fatal error to exit 2.
|
// maps everything that is not a fatal error to exit 2.
|
||||||
@@ -143,11 +122,6 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
|||||||
root.SetErr(stderr)
|
root.SetErr(stderr)
|
||||||
root.CompletionOptions.DisableDefaultCmd = true
|
root.CompletionOptions.DisableDefaultCmd = true
|
||||||
|
|
||||||
// Cobra's built-in version flag prints through the help writer,
|
|
||||||
// stderr; this one prints to stdout.
|
|
||||||
root.Flags().BoolVarP(&showVersion, "version", "v", false,
|
|
||||||
"print the version to stdout")
|
|
||||||
|
|
||||||
var (
|
var (
|
||||||
scanWorkers int
|
scanWorkers int
|
||||||
scanOneFS bool
|
scanOneFS bool
|
||||||
@@ -157,13 +131,7 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
|||||||
Use: cmdScan + " [--workers N] [-x] PATH...",
|
Use: cmdScan + " [--workers N] [-x] PATH...",
|
||||||
Short: "Walk trees and synchronize the scan database",
|
Short: "Walk trees and synchronize the scan database",
|
||||||
Args: cobra.MinimumNArgs(1),
|
Args: cobra.MinimumNArgs(1),
|
||||||
PreRunE: func(cmd *cobra.Command, _ []string) error {
|
|
||||||
return checkScanWorkers(cmd, scanWorkers)
|
|
||||||
},
|
|
||||||
RunE: runE(func(ctx context.Context, args []string) error {
|
RunE: runE(func(ctx context.Context, args []string) error {
|
||||||
ctx, stop := interruptContext(ctx)
|
|
||||||
defer stop()
|
|
||||||
|
|
||||||
return runScan(ctx, args, scanWorkers, scanOneFS)
|
return runScan(ctx, args, scanWorkers, scanOneFS)
|
||||||
}),
|
}),
|
||||||
}
|
}
|
||||||
@@ -195,26 +163,13 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
|||||||
return root
|
return root
|
||||||
}
|
}
|
||||||
|
|
||||||
// checkScanWorkers rejects a scan --workers value below 1. That is a
|
// runE adapts a subcommand implementation to cobra's RunE. Cobra
|
||||||
// usage error reported in one line: cobra prints the returned message
|
// prints the error and the command's usage text for every error RunE
|
||||||
// without the usage text, and run exits 2.
|
// returns, but a subcommand that ran and failed has no usage problem
|
||||||
func checkScanWorkers(cmd *cobra.Command, workers int) error {
|
// to report: both are silenced here, and the error is marked fatal so
|
||||||
if workers >= 1 {
|
// that run reports it on stderr and exits 1 rather than 2. The command's
|
||||||
return nil
|
// context is handed to the implementation: cancelling it unwinds the
|
||||||
}
|
// scan's worker pools.
|
||||||
|
|
||||||
cmd.SilenceUsage = true
|
|
||||||
|
|
||||||
return fmt.Errorf("%w, got %d", errWorkersBelowOne, workers)
|
|
||||||
}
|
|
||||||
|
|
||||||
// runE adapts a subcommand implementation, or the version print, to
|
|
||||||
// cobra's RunE. Cobra prints the error and the command's usage text for
|
|
||||||
// every error RunE returns, but a subcommand that ran and failed has no
|
|
||||||
// usage problem to report: both are silenced here, and the error is
|
|
||||||
// marked fatal so that run reports it on stderr and exits 1 rather than
|
|
||||||
// 2. The command's context is handed to the implementation: cancelling
|
|
||||||
// it unwinds the scan's worker pools.
|
|
||||||
func runE(
|
func runE(
|
||||||
fn func(ctx context.Context, args []string) error,
|
fn func(ctx context.Context, args []string) error,
|
||||||
) func(*cobra.Command, []string) error {
|
) func(*cobra.Command, []string) error {
|
||||||
|
|||||||
+20
-162
@@ -8,7 +8,6 @@ import (
|
|||||||
"io/fs"
|
"io/fs"
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"slices"
|
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
@@ -296,108 +295,34 @@ func TestRunUsageErrors(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRunScanRejectsWorkersBelowOne(t *testing.T) {
|
// TestRunHelpAndVersionSucceed checks that the two informational flags
|
||||||
// README §scan mode: --workers below 1 is a usage error reported in
|
// exit 0 and keep their human-facing output on stderr.
|
||||||
// one line on stderr, before the scan opens the database.
|
func TestRunHelpAndVersionSucceed(t *testing.T) {
|
||||||
for _, workers := range []string{"0", "-1"} {
|
|
||||||
t.Run(workers, func(t *testing.T) {
|
|
||||||
dbPath := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, dbPath)
|
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
args := []string{cmdScan, "--workers", workers, t.TempDir()}
|
|
||||||
|
|
||||||
code := run(args, &stdout, &stderr)
|
|
||||||
if code != exitUsage {
|
|
||||||
t.Errorf("run(%v) = %d, want %d", args, code, exitUsage)
|
|
||||||
}
|
|
||||||
|
|
||||||
want := "Error: --workers must be at least 1, got " + workers +
|
|
||||||
"\n"
|
|
||||||
if got := stderr.String(); got != want {
|
|
||||||
t.Errorf("stderr = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
if got := stdout.String(); got != "" {
|
|
||||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err := os.Stat(dbPath)
|
|
||||||
if !errors.Is(err, fs.ErrNotExist) {
|
|
||||||
t.Errorf("stat %s: %v, want the database never created",
|
|
||||||
dbPath, err)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunHelp(t *testing.T) {
|
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// README §Subcommands: help goes to stderr, exits 0, and leaves
|
assertHumanOutput(t, "--help")
|
||||||
// stdout empty.
|
assertHumanOutput(t, "--version")
|
||||||
cases := [][]string{{"--help"}, {"-h"}, {cmdScan, "--help"}}
|
|
||||||
|
|
||||||
for _, args := range cases {
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run(args, &stdout, &stderr)
|
|
||||||
if code != exitOK {
|
|
||||||
t.Errorf("run(%v) = %d, want %d", args, code, exitOK)
|
|
||||||
}
|
|
||||||
|
|
||||||
if !strings.Contains(stderr.String(), usageMarker) {
|
|
||||||
t.Errorf("run(%v) stderr = %q, want the help text",
|
|
||||||
args, stderr.String())
|
|
||||||
}
|
|
||||||
|
|
||||||
if got := stdout.String(); got != "" {
|
|
||||||
t.Errorf("run(%v) stdout = %q, want nothing (data only)",
|
|
||||||
args, got)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRunVersion(t *testing.T) {
|
// assertHumanOutput runs sfdupes with one informational flag and checks
|
||||||
t.Parallel()
|
// that it succeeds with its output on stderr and stdout untouched
|
||||||
|
// (README design goal 4).
|
||||||
|
func assertHumanOutput(t *testing.T, arg string) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
// README §Subcommands: the version is one line on stdout, with
|
var stdout, stderr bytes.Buffer
|
||||||
// nothing on stderr, and exits 0.
|
|
||||||
for _, arg := range []string{"--version", "-v"} {
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{arg}, &stdout, &stderr)
|
code := run([]string{arg}, &stdout, &stderr)
|
||||||
if code != exitOK {
|
if code != exitOK {
|
||||||
t.Errorf("run(%s) = %d, want %d", arg, code, exitOK)
|
t.Errorf("run(%s) = %d, want %d", arg, code, exitOK)
|
||||||
}
|
|
||||||
|
|
||||||
want := "sfdupes " + Version + "\n"
|
|
||||||
if got := stdout.String(); got != want {
|
|
||||||
t.Errorf("run(%s) stdout = %q, want %q", arg, got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
if got := stderr.String(); got != "" {
|
|
||||||
t.Errorf("run(%s) stderr = %q, want nothing", arg, got)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunVersionWriteFailureIsFatal(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
// README §Error handling: a stdout write failure exits 1, reported
|
|
||||||
// in one line on stderr.
|
|
||||||
var stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{"--version"}, failingWriter{}, &stderr)
|
|
||||||
if code != exitFatal {
|
|
||||||
t.Errorf("run(--version) = %d, want %d", code, exitFatal)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
want := "sfdupes: write stdout: " + errWriteFailed.Error() + "\n"
|
if stderr.Len() == 0 {
|
||||||
if got := stderr.String(); got != want {
|
t.Errorf("run(%s) wrote nothing to stderr", arg)
|
||||||
t.Errorf("stderr = %q, want %q", got, want)
|
}
|
||||||
|
|
||||||
|
if got := stdout.String(); got != "" {
|
||||||
|
t.Errorf("stdout = %q, want nothing (data only)", got)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -640,73 +565,6 @@ func TestRunReportsNeedOnlyReadAccess(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// assertRunsUseDatabase runs scan, then report and trees, against the
|
|
||||||
// database that SFDUPES_DATABASE names, the file name in dir. It fails
|
|
||||||
// unless the reports find the duplicate pair the scan recorded and dir
|
|
||||||
// then holds only that file and its lock file: nothing was created
|
|
||||||
// under a shortened name.
|
|
||||||
func assertRunsUseDatabase(t *testing.T, dir, name string) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
dupes := scanFixture(t)
|
|
||||||
|
|
||||||
want := "first\tdupe\tsize\n" + dupes[0] + "\t" + dupes[1] + "\t300\n"
|
|
||||||
if got := runStdout(t, cmdReport); got != want {
|
|
||||||
t.Errorf("report stdout = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
want = "first\tdupe\tfiles\tsize\n" +
|
|
||||||
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) + "\t1\t300\n"
|
|
||||||
if got := runStdout(t, cmdTrees); got != want {
|
|
||||||
t.Errorf("trees stdout = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
entries, err := os.ReadDir(dir)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
got := make([]string, 0, len(entries))
|
|
||||||
for _, e := range entries {
|
|
||||||
got = append(got, e.Name())
|
|
||||||
}
|
|
||||||
|
|
||||||
if wantFiles := []string{name, name + ".lock"}; !slices.Equal(got, wantFiles) {
|
|
||||||
t.Errorf("%s holds %q, want %q", dir, got, wantFiles)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunDatabasePathUsedAsGiven(t *testing.T) {
|
|
||||||
// README §Database: the path names the database file exactly. In
|
|
||||||
// SQLite's connection string an unescaped ? or # would end the file
|
|
||||||
// name and % would start an escape, and a path starting with //
|
|
||||||
// could be read as a host name. %25 is a valid escape, so unescaped
|
|
||||||
// this name opens a file named a without any error.
|
|
||||||
const name = "a?b#c%25d e.sqlite"
|
|
||||||
|
|
||||||
t.Run("absolute", func(t *testing.T) {
|
|
||||||
dir := t.TempDir()
|
|
||||||
t.Setenv(databaseEnv, filepath.Join(dir, name))
|
|
||||||
|
|
||||||
assertRunsUseDatabase(t, dir, name)
|
|
||||||
})
|
|
||||||
|
|
||||||
t.Run("leading double slash", func(t *testing.T) {
|
|
||||||
dir := t.TempDir()
|
|
||||||
t.Setenv(databaseEnv, "/"+filepath.Join(dir, name))
|
|
||||||
|
|
||||||
assertRunsUseDatabase(t, dir, name)
|
|
||||||
})
|
|
||||||
|
|
||||||
t.Run("relative", func(t *testing.T) {
|
|
||||||
dir := t.TempDir()
|
|
||||||
t.Chdir(dir)
|
|
||||||
t.Setenv(databaseEnv, name)
|
|
||||||
|
|
||||||
assertRunsUseDatabase(t, dir, name)
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
// holdScanLock takes the lock on the database at path, as a running
|
// holdScanLock takes the lock on the database at path, as a running
|
||||||
// scan does, and holds it until the test ends. It fails the test when
|
// scan does, and holds it until the test ends. It fails the test when
|
||||||
// the lock is already held.
|
// the lock is already held.
|
||||||
|
|||||||
+2
-8
@@ -141,20 +141,14 @@ func (p *progress) warnf(format string, args ...any) {
|
|||||||
fmt.Fprintln(os.Stderr, msg)
|
fmt.Fprintln(os.Stderr, msg)
|
||||||
}
|
}
|
||||||
|
|
||||||
// finish terminates the pass's display. A bar whose pass stopped short
|
// finish terminates the pass's display.
|
||||||
// of its total, as an interrupted one does, is left as last drawn; the
|
|
||||||
// library's Finish would fill it up.
|
|
||||||
func (p *progress) finish() {
|
func (p *progress) finish() {
|
||||||
if p == nil {
|
if p == nil {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
if p.bar != nil {
|
if p.bar != nil {
|
||||||
if p.total >= 0 && p.count < p.total {
|
_ = p.bar.Finish()
|
||||||
_ = p.bar.Exit()
|
|
||||||
} else {
|
|
||||||
_ = p.bar.Finish()
|
|
||||||
}
|
|
||||||
|
|
||||||
fmt.Fprintln(os.Stderr)
|
fmt.Fprintln(os.Stderr)
|
||||||
|
|
||||||
|
|||||||
+6
-34
@@ -144,46 +144,18 @@ func TestSpinnerShowsCountAfterBurst(t *testing.T) {
|
|||||||
|
|
||||||
time.Sleep(spinnerIdle)
|
time.Sleep(spinnerIdle)
|
||||||
|
|
||||||
if shown := lastFrame(stderr()); !strings.Contains(shown, "(50/-,") {
|
// A terminal shows the last frame drawn. The library starts each
|
||||||
t.Errorf("terminal shows %q, want a count of 50", shown)
|
// frame with a carriage return and erases the previous one with
|
||||||
}
|
// spaces first.
|
||||||
}
|
|
||||||
|
|
||||||
// lastFrame returns what a terminal shows of the frames a bar drew: the
|
|
||||||
// last one. The library starts each frame with a carriage return and
|
|
||||||
// erases the previous one with spaces first.
|
|
||||||
func lastFrame(out string) string {
|
|
||||||
var shown string
|
var shown string
|
||||||
|
|
||||||
for frame := range strings.SplitSeq(out, "\r") {
|
for frame := range strings.SplitSeq(stderr(), "\r") {
|
||||||
if strings.TrimSpace(frame) != "" {
|
if strings.TrimSpace(frame) != "" {
|
||||||
shown = frame
|
shown = frame
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return shown
|
if !strings.Contains(shown, "(50/-,") {
|
||||||
}
|
t.Errorf("terminal shows %q, want a count of 50", shown)
|
||||||
|
|
||||||
// TestBarStoppedShortKeepsCount checks that the terminal display of a
|
|
||||||
// pass that stops before its total, as an interrupted one does, is left
|
|
||||||
// as last drawn instead of being filled up.
|
|
||||||
//
|
|
||||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
|
||||||
func TestBarStoppedShortKeepsCount(t *testing.T) {
|
|
||||||
stderr := captureStderr(t)
|
|
||||||
p := &progress{
|
|
||||||
label: "hash", total: 10, start: time.Now(),
|
|
||||||
bar: newBar("hash", 10),
|
|
||||||
}
|
|
||||||
|
|
||||||
p.increment()
|
|
||||||
|
|
||||||
// Past the redraw limit, so the bar draws the next count.
|
|
||||||
time.Sleep(2 * barThrottle)
|
|
||||||
p.increment()
|
|
||||||
p.finish()
|
|
||||||
|
|
||||||
if shown := lastFrame(stderr()); !strings.Contains(shown, "(2/10,") {
|
|
||||||
t.Errorf("terminal shows %q, want a count of 2 of 10", shown)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+4
-4
@@ -122,12 +122,12 @@ func TestReportStdoutFailsWhileReading(t *testing.T) {
|
|||||||
// still being read, not at the final flush.
|
// still being read, not at the final flush.
|
||||||
dir := "/" + strings.Repeat("d", 4096)
|
dir := "/" + strings.Repeat("d", 4096)
|
||||||
|
|
||||||
recs := make([]scanRec, ioBufSize/len(dir))
|
var recs []scanRec
|
||||||
for i := range recs {
|
for i := range ioBufSize / len(dir) {
|
||||||
recs[i] = scanRec{
|
recs = append(recs, scanRec{
|
||||||
size: 1, head: "h", tail: "t", content: "c",
|
size: 1, head: "h", tail: "t", content: "c",
|
||||||
path: fmt.Sprintf("%s/%d", dir, i),
|
path: fmt.Sprintf("%s/%d", dir, i),
|
||||||
}
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, recs))
|
t.Setenv(databaseEnv, seedDatabase(t, recs))
|
||||||
|
|||||||
@@ -11,7 +11,6 @@ import (
|
|||||||
"io"
|
"io"
|
||||||
"io/fs"
|
"io/fs"
|
||||||
"os"
|
"os"
|
||||||
"os/signal"
|
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"slices"
|
"slices"
|
||||||
"strings"
|
"strings"
|
||||||
@@ -58,10 +57,6 @@ const sampleWindow = 1024 * 1024
|
|||||||
// and hash worker pools.
|
// and hash worker pools.
|
||||||
const workQueueDepth = 1024
|
const workQueueDepth = 1024
|
||||||
|
|
||||||
// errInterrupted reports a scan stopped by SIGINT or SIGTERM. runScan
|
|
||||||
// has already printed its line, so run prints nothing more.
|
|
||||||
var errInterrupted = errors.New("scan interrupted")
|
|
||||||
|
|
||||||
// fileRec carries one statted file between the scan phases. dev and
|
// fileRec carries one statted file between the scan phases. dev and
|
||||||
// ino identify the underlying inode so hard-linked paths can share
|
// ino identify the underlying inode so hard-linked paths can share
|
||||||
// one read; both are zero when the platform exposes no inode.
|
// one read; both are zero when the platform exposes no inode.
|
||||||
@@ -95,14 +90,15 @@ type fileMeta struct {
|
|||||||
// scan fails before it walks the filesystem or opens the database.
|
// scan fails before it walks the filesystem or opens the database.
|
||||||
// Errors are returned rather than exiting, so that the deferred close —
|
// Errors are returned rather than exiting, so that the deferred close —
|
||||||
// which takes the database out of WAL mode — always runs, and the lock
|
// which takes the database out of WAL mode — always runs, and the lock
|
||||||
// is released after it. When ctx is cancelled, as by the SIGINT or
|
// is released after it. Cancelling ctx unwinds the worker pools and
|
||||||
// SIGTERM that interruptContext catches, the scan keeps what it has
|
// aborts the scan with the context's error.
|
||||||
// hashed (see syncScan), prints how many files its walk reached, and
|
|
||||||
// returns errInterrupted. workers must be at least 1; the scan command
|
|
||||||
// rejects anything less.
|
|
||||||
func runScan(ctx context.Context, roots []string, workers int,
|
func runScan(ctx context.Context, roots []string, workers int,
|
||||||
oneFS bool,
|
oneFS bool,
|
||||||
) error {
|
) error {
|
||||||
|
if workers < 1 {
|
||||||
|
workers = 1
|
||||||
|
}
|
||||||
|
|
||||||
roots, err := resolveRoots(roots)
|
roots, err := resolveRoots(roots)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
@@ -118,12 +114,6 @@ func runScan(ctx context.Context, roots []string, workers int,
|
|||||||
defer func() { _ = lock.Close() }()
|
defer func() { _ = lock.Close() }()
|
||||||
|
|
||||||
db, err := openScanDatabase(ctx, dbPath)
|
db, err := openScanDatabase(ctx, dbPath)
|
||||||
if err != nil && ctx.Err() != nil {
|
|
||||||
// Interrupted while opening; SQLite may report that with an
|
|
||||||
// error of its own rather than the context's.
|
|
||||||
return interrupted(0)
|
|
||||||
}
|
|
||||||
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
@@ -131,10 +121,6 @@ func runScan(ctx context.Context, roots []string, workers int,
|
|||||||
defer closeScanDatabase(ctx, db, dbPath)
|
defer closeScanDatabase(ctx, db, dbPath)
|
||||||
|
|
||||||
st, err := syncScan(ctx, db, roots, workers, oneFS)
|
st, err := syncScan(ctx, db, roots, workers, oneFS)
|
||||||
if errors.Is(err, context.Canceled) {
|
|
||||||
return interrupted(st.walked)
|
|
||||||
}
|
|
||||||
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("update database %s: %w", dbPath, err)
|
return fmt.Errorf("update database %s: %w", dbPath, err)
|
||||||
}
|
}
|
||||||
@@ -148,34 +134,6 @@ func runScan(ctx context.Context, roots []string, workers int,
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// interruptContext returns a copy of ctx that the first SIGINT or
|
|
||||||
// SIGTERM cancels; the scan command runs the scan under it. stop
|
|
||||||
// releases the signals.
|
|
||||||
func interruptContext(ctx context.Context) (context.Context, func()) {
|
|
||||||
// A SIGINT ignored from the start, as by a script's background job,
|
|
||||||
// stays ignored.
|
|
||||||
signals := []os.Signal{syscall.SIGTERM}
|
|
||||||
if !signal.Ignored(syscall.SIGINT) {
|
|
||||||
signals = append(signals, syscall.SIGINT)
|
|
||||||
}
|
|
||||||
|
|
||||||
ctx, stop := signal.NotifyContext(ctx, signals...)
|
|
||||||
|
|
||||||
// Stopping restores the default handling, so a second signal ends
|
|
||||||
// the process at once.
|
|
||||||
context.AfterFunc(ctx, stop)
|
|
||||||
|
|
||||||
return ctx, stop
|
|
||||||
}
|
|
||||||
|
|
||||||
// interrupted prints the line for a scan stopped by a signal after its
|
|
||||||
// walk reached walked files, and returns errInterrupted.
|
|
||||||
func interrupted(walked int) error {
|
|
||||||
fmt.Fprintf(os.Stderr, "scan: interrupted after %d files\n", walked)
|
|
||||||
|
|
||||||
return errInterrupted
|
|
||||||
}
|
|
||||||
|
|
||||||
// resolveRoots converts each PATH operand to an absolute, lexically
|
// resolveRoots converts each PATH operand to an absolute, lexically
|
||||||
// cleaned path (symlinks are not resolved) and verifies that it
|
// cleaned path (symlinks are not resolved) and verifies that it
|
||||||
// exists. Database records are keyed by absolute path, so scan results
|
// exists. Database records are keyed by absolute path, so scan results
|
||||||
@@ -229,9 +187,8 @@ func pruneRoots(roots []string) []string {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// scanStats summarizes one scan's database synchronization for the
|
// scanStats summarizes one scan's database synchronization for the
|
||||||
// final stderr summary, or for the line an interrupted scan prints.
|
// final stderr summary.
|
||||||
type scanStats struct {
|
type scanStats struct {
|
||||||
walked int // files the walk reached
|
|
||||||
added int
|
added int
|
||||||
updated int
|
updated int
|
||||||
removed int
|
removed int
|
||||||
@@ -253,30 +210,7 @@ type scanState struct {
|
|||||||
st scanStats
|
st scanStats
|
||||||
}
|
}
|
||||||
|
|
||||||
// syncScan synchronizes the database with the filesystem under roots;
|
// syncScan synchronizes the database with the filesystem under roots
|
||||||
// see runPhases. When ctx is cancelled, as by an interrupt, it commits
|
|
||||||
// the hashed records still waiting in the batch, starts no other write
|
|
||||||
// or deletion, and returns the cancellation.
|
|
||||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
|
||||||
workers int, oneFS bool,
|
|
||||||
) (scanStats, error) {
|
|
||||||
s := &scanState{db: db}
|
|
||||||
|
|
||||||
err := s.runPhases(ctx, roots, workers, oneFS)
|
|
||||||
if err == nil || ctx.Err() == nil {
|
|
||||||
return s.st, err
|
|
||||||
}
|
|
||||||
|
|
||||||
// The one write made after the cancellation, so it cannot use ctx.
|
|
||||||
err = applyChanges(context.WithoutCancel(ctx), db, s.batch, nil, nil)
|
|
||||||
if err != nil {
|
|
||||||
return s.st, err
|
|
||||||
}
|
|
||||||
|
|
||||||
return s.st, ctx.Err()
|
|
||||||
}
|
|
||||||
|
|
||||||
// runPhases synchronizes the database with the filesystem under roots
|
|
||||||
// in four sequential phases: walk (enumerate and stat every file,
|
// in four sequential phases: walk (enumerate and stat every file,
|
||||||
// building a complete size census), hash (read only the new or
|
// building a complete size census), hash (read only the new or
|
||||||
// changed — or previously unhashed — files whose size at least one
|
// changed — or previously unhashed — files whose size at least one
|
||||||
@@ -289,41 +223,48 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
|||||||
// their content hash. Operands the walk cannot start from are dropped
|
// their content hash. Operands the walk cannot start from are dropped
|
||||||
// first, so the records beneath them count as outside the roots unless
|
// first, so the records beneath them count as outside the roots unless
|
||||||
// they lie under another root.
|
// they lie under another root.
|
||||||
func (s *scanState) runPhases(ctx context.Context, roots []string,
|
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||||
workers int, oneFS bool,
|
workers int, oneFS bool,
|
||||||
) error {
|
) (scanStats, error) {
|
||||||
|
s := &scanState{db: db}
|
||||||
|
|
||||||
// Types are checked before pruning so that an operand under a
|
// Types are checked before pruning so that an operand under a
|
||||||
// dropped one is still scanned, not dropped as lying under it.
|
// dropped one is still scanned, not dropped as lying under it.
|
||||||
roots = pruneRoots(s.walkableRoots(roots))
|
roots = pruneRoots(s.walkableRoots(roots))
|
||||||
|
|
||||||
err := s.loadIndex(ctx, roots)
|
err := s.loadIndex(ctx, roots)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return s.st, err
|
||||||
}
|
}
|
||||||
|
|
||||||
changed, unhashed := s.walkPhase(startWalk(ctx, roots, oneFS, workers))
|
changed, unhashed := s.walkPhase(startWalk(ctx, roots, oneFS, workers))
|
||||||
|
|
||||||
// A cancelled walk stops early, so its size census covers only part
|
// A cancelled walk stops early, so its size census covers only part
|
||||||
// of the roots, and every file it never reached would look vanished
|
// of the roots, and every file it never reached looks vanished to
|
||||||
// to the update phase. Stop before anything is written or deleted.
|
// the update phase. Defence in depth rather than the only barrier:
|
||||||
|
// that phase would today fail on its first BeginTx with the same
|
||||||
|
// cancelled context before deleting anything. But it is the barrier
|
||||||
|
// that survives a later decision to let an interrupted scan commit
|
||||||
|
// what it has, and it turns a confusing failure deep in the update
|
||||||
|
// phase into a clean abort at the phase boundary.
|
||||||
err = ctx.Err()
|
err = ctx.Err()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return s.st, err
|
||||||
}
|
}
|
||||||
|
|
||||||
s.partition(changed, unhashed)
|
s.partition(changed, unhashed)
|
||||||
|
|
||||||
err = s.hashPhase(ctx, workers)
|
err = s.hashPhase(ctx, workers)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return s.st, err
|
||||||
}
|
}
|
||||||
|
|
||||||
err = s.updatePhase(ctx)
|
err = s.updatePhase(ctx)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return s.st, err
|
||||||
}
|
}
|
||||||
|
|
||||||
return s.contentPhase(ctx, workers)
|
return s.st, s.contentPhase(ctx, workers)
|
||||||
}
|
}
|
||||||
|
|
||||||
// walkableRoots returns the operands the walk can start from: regular
|
// walkableRoots returns the operands the walk can start from: regular
|
||||||
@@ -408,7 +349,6 @@ func (s *scanState) walkPhase(
|
|||||||
}
|
}
|
||||||
|
|
||||||
s.sizes = append(s.sizes, ev.rec.size)
|
s.sizes = append(s.sizes, ev.rec.size)
|
||||||
s.st.walked++
|
|
||||||
|
|
||||||
prog.increment()
|
prog.increment()
|
||||||
|
|
||||||
@@ -612,22 +552,16 @@ func (s *scanState) recordRun(ctx context.Context, r hashResult) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// commitFullBatch commits the running batch once it holds
|
// commitFullBatch commits the running batch once it holds
|
||||||
// updateBatchSize records. A batch that fails to commit is kept: the
|
// updateBatchSize records.
|
||||||
// commit fails when the scan is interrupted, and syncScan then commits
|
|
||||||
// the batch itself.
|
|
||||||
func (s *scanState) commitFullBatch(ctx context.Context) error {
|
func (s *scanState) commitFullBatch(ctx context.Context) error {
|
||||||
if len(s.batch) < updateBatchSize {
|
if len(s.batch) < updateBatchSize {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
err := applyBatch(ctx, s.db, s.batch, nil, nil)
|
err := applyBatch(ctx, s.db, s.batch, nil, nil)
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
|
|
||||||
s.batch = s.batch[:0]
|
s.batch = s.batch[:0]
|
||||||
|
|
||||||
return nil
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
// updatePhase writes the scan's tail under one progress display: the
|
// updatePhase writes the scan's tail under one progress display: the
|
||||||
@@ -1070,12 +1004,6 @@ func walkOneDir(ctx context.Context, job dirJob, oneFS bool,
|
|||||||
var subs []dirJob
|
var subs []dirJob
|
||||||
|
|
||||||
for _, e := range entries {
|
for _, e := range entries {
|
||||||
// A cancelled scan wants nothing more from this directory: stop
|
|
||||||
// rather than lstat the rest of a large one.
|
|
||||||
if ctx.Err() != nil {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
p := filepath.Join(job.path, e.Name())
|
p := filepath.Join(job.path, e.Name())
|
||||||
|
|
||||||
if e.IsDir() {
|
if e.IsDir() {
|
||||||
|
|||||||
+16
-180
@@ -6,18 +6,14 @@ import (
|
|||||||
"crypto/sha256"
|
"crypto/sha256"
|
||||||
"database/sql"
|
"database/sql"
|
||||||
"encoding/hex"
|
"encoding/hex"
|
||||||
"errors"
|
|
||||||
"fmt"
|
"fmt"
|
||||||
"io"
|
"io"
|
||||||
"io/fs"
|
|
||||||
"os"
|
"os"
|
||||||
"os/signal"
|
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"runtime"
|
"runtime"
|
||||||
"slices"
|
"slices"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
"syscall"
|
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
"time"
|
||||||
)
|
)
|
||||||
@@ -460,7 +456,7 @@ func TestScanContentWithinOperand(t *testing.T) {
|
|||||||
added := sparseFile(t, dir, "d2", headTailMin)
|
added := sparseFile(t, dir, "d2", headTailMin)
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 3, added: 1, unchanged: 2}) {
|
if st != (scanStats{added: 1, unchanged: 2}) {
|
||||||
t.Fatalf("rescan stats = %+v, want 1 added 2 unchanged", st)
|
t.Fatalf("rescan stats = %+v, want 1 added 2 unchanged", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -506,7 +502,7 @@ func TestScanContentStalePartners(t *testing.T) {
|
|||||||
sparseFile(t, dirB, "changed-copy", headTailMin+1)
|
sparseFile(t, dirB, "changed-copy", headTailMin+1)
|
||||||
|
|
||||||
st := syncTree(t, db, dirB)
|
st := syncTree(t, db, dirB)
|
||||||
if st != (scanStats{walked: 2, added: 2}) {
|
if st != (scanStats{added: 2}) {
|
||||||
t.Errorf("stats = %+v, want 2 added and nothing skipped", st)
|
t.Errorf("stats = %+v, want 2 added and nothing skipped", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -563,7 +559,7 @@ func TestScanContentHashedStalePartners(t *testing.T) {
|
|||||||
b := sparseFile(t, t.TempDir(), "copy", headTailMin)
|
b := sparseFile(t, t.TempDir(), "copy", headTailMin)
|
||||||
|
|
||||||
st := syncTree(t, db, filepath.Dir(b))
|
st := syncTree(t, db, filepath.Dir(b))
|
||||||
if st != (scanStats{walked: 1, added: 1}) {
|
if st != (scanStats{added: 1}) {
|
||||||
t.Errorf("stats = %+v, want 1 added and nothing skipped", st)
|
t.Errorf("stats = %+v, want 1 added and nothing skipped", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -603,7 +599,7 @@ func TestScanContentReadFailure(t *testing.T) {
|
|||||||
b := sparseFile(t, dirB, "b", headTailMin)
|
b := sparseFile(t, dirB, "b", headTailMin)
|
||||||
|
|
||||||
st := syncTree(t, db, dirB)
|
st := syncTree(t, db, dirB)
|
||||||
if st != (scanStats{walked: 1, added: 1, skipped: 1}) {
|
if st != (scanStats{added: 1, skipped: 1}) {
|
||||||
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -617,7 +613,7 @@ func TestScanContentReadFailure(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
st = syncTree(t, db, dirB)
|
st = syncTree(t, db, dirB)
|
||||||
if st != (scanStats{walked: 1, unchanged: 1}) {
|
if st != (scanStats{unchanged: 1}) {
|
||||||
t.Fatalf("rescan stats = %+v, want 1 unchanged", st)
|
t.Fatalf("rescan stats = %+v, want 1 unchanged", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -663,7 +659,7 @@ func TestScanContentCheckError(t *testing.T) {
|
|||||||
b := sparseFile(t, t.TempDir(), "b", headTailMin)
|
b := sparseFile(t, t.TempDir(), "b", headTailMin)
|
||||||
|
|
||||||
st := syncTree(t, db, filepath.Dir(b))
|
st := syncTree(t, db, filepath.Dir(b))
|
||||||
if st != (scanStats{walked: 1, added: 1, skipped: 1}) {
|
if st != (scanStats{added: 1, skipped: 1}) {
|
||||||
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
t.Fatalf("stats = %+v, want 1 added 1 skipped", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -691,7 +687,7 @@ func TestScanContentHardlinks(t *testing.T) {
|
|||||||
c := sparseFile(t, dir, "copy", headTailMin)
|
c := sparseFile(t, dir, "copy", headTailMin)
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 3, added: 3}) {
|
if st != (scanStats{added: 3}) {
|
||||||
t.Fatalf("stats = %+v, want 3 added", st)
|
t.Fatalf("stats = %+v, want 3 added", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -914,159 +910,6 @@ func TestDeviceOfInfo(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// dirEntryFor returns the fs.DirEntry for name within dir, obtained via
|
|
||||||
// the same os.ReadDir the walk uses, so it carries a real Info().
|
|
||||||
func dirEntryFor(t *testing.T, dir, name string) fs.DirEntry {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
entries, err := os.ReadDir(dir)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
for _, e := range entries {
|
|
||||||
if e.Name() == name {
|
|
||||||
return e
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
t.Fatalf("entry %q not found in %q", name, dir)
|
|
||||||
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// callSubdirJob runs subdirJob against e under parent, collecting any
|
|
||||||
// warning events it emits (subdirJob emits at most one).
|
|
||||||
func callSubdirJob(t *testing.T, p string, e fs.DirEntry,
|
|
||||||
parent dirJob, oneFS bool,
|
|
||||||
) (dirJob, bool, []walkEvent) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
events := make(chan walkEvent, 1)
|
|
||||||
job, ok := subdirJob(t.Context(), p, e, parent, oneFS, events)
|
|
||||||
close(events)
|
|
||||||
|
|
||||||
var evs []walkEvent
|
|
||||||
for ev := range events {
|
|
||||||
evs = append(evs, ev)
|
|
||||||
}
|
|
||||||
|
|
||||||
return job, ok, evs
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestSubdirJobOneFilesystem exercises the -x boundary check in
|
|
||||||
// subdirJob directly, so no second real filesystem is needed. The
|
|
||||||
// subdirectory's real device is compared against a fabricated operand
|
|
||||||
// device.
|
|
||||||
func TestSubdirJobOneFilesystem(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
dir := t.TempDir()
|
|
||||||
|
|
||||||
sub := filepath.Join(dir, "sub")
|
|
||||||
|
|
||||||
err := os.Mkdir(sub, 0o750)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
info, err := os.Lstat(sub)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
dev, ok := deviceOfInfo(info)
|
|
||||||
if !ok {
|
|
||||||
t.Skip("platform exposes no device id")
|
|
||||||
}
|
|
||||||
|
|
||||||
// A device the subdirectory is not on, standing in for an operand
|
|
||||||
// rooted on a different filesystem.
|
|
||||||
otherDev := dev + 1
|
|
||||||
e := dirEntryFor(t, dir, "sub")
|
|
||||||
|
|
||||||
cases := []struct {
|
|
||||||
name string
|
|
||||||
oneFS bool
|
|
||||||
parent dirJob
|
|
||||||
wantOK bool
|
|
||||||
}{
|
|
||||||
// -x on, subdirectory on a different device than its operand:
|
|
||||||
// descent is refused.
|
|
||||||
{"reject across boundary", true,
|
|
||||||
dirJob{rootDev: otherDev, rootDevOK: true}, false},
|
|
||||||
// -x on but the operand's own device is unknown: the boundary
|
|
||||||
// check is bypassed and descent proceeds.
|
|
||||||
{"bypass when root device unknown", true,
|
|
||||||
dirJob{rootDev: otherDev, rootDevOK: false}, true},
|
|
||||||
// Default (no -x): boundaries are crossed even onto a different
|
|
||||||
// device.
|
|
||||||
{"cross by default", false,
|
|
||||||
dirJob{rootDev: otherDev, rootDevOK: true}, true},
|
|
||||||
}
|
|
||||||
|
|
||||||
for _, tc := range cases {
|
|
||||||
t.Run(tc.name, func(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
job, ok, evs := callSubdirJob(t, sub, e, tc.parent, tc.oneFS)
|
|
||||||
if ok != tc.wantOK {
|
|
||||||
t.Fatalf("accepted = %v, want %v", ok, tc.wantOK)
|
|
||||||
}
|
|
||||||
|
|
||||||
if len(evs) != 0 {
|
|
||||||
t.Fatalf("unexpected events: %+v", evs)
|
|
||||||
}
|
|
||||||
|
|
||||||
// The accepted job must carry the operand's device down, or -x
|
|
||||||
// stops checking below the first level.
|
|
||||||
want := dirJob{
|
|
||||||
path: sub,
|
|
||||||
rootDev: tc.parent.rootDev,
|
|
||||||
rootDevOK: tc.parent.rootDevOK,
|
|
||||||
}
|
|
||||||
if ok && job != want {
|
|
||||||
t.Fatalf("job = %+v, want %+v", job, want)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// errInfoUnavailable is returned by errDirEntry.Info().
|
|
||||||
var errInfoUnavailable = errors.New("info unavailable")
|
|
||||||
|
|
||||||
// errDirEntry is a directory entry whose Info() always fails, driving
|
|
||||||
// subdirJob's stat-error branch deterministically.
|
|
||||||
type errDirEntry struct{ name string }
|
|
||||||
|
|
||||||
func (e errDirEntry) Name() string { return e.name }
|
|
||||||
func (errDirEntry) IsDir() bool { return true }
|
|
||||||
func (errDirEntry) Type() fs.FileMode {
|
|
||||||
return fs.ModeDir
|
|
||||||
}
|
|
||||||
|
|
||||||
func (errDirEntry) Info() (fs.FileInfo, error) {
|
|
||||||
return nil, errInfoUnavailable
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestSubdirJobStatError asserts that when a subdirectory's Info()
|
|
||||||
// fails under -x, subdirJob warns and refuses descent.
|
|
||||||
func TestSubdirJobStatError(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
p := "/does/not/matter/sub"
|
|
||||||
|
|
||||||
_, ok, evs := callSubdirJob(t, p, errDirEntry{name: "sub"},
|
|
||||||
dirJob{rootDev: 1, rootDevOK: true}, true)
|
|
||||||
if ok {
|
|
||||||
t.Fatal("descent accepted after stat error, want refused")
|
|
||||||
}
|
|
||||||
|
|
||||||
if len(evs) != 1 || !evs[0].fail || !strings.Contains(evs[0].warn, p) {
|
|
||||||
t.Fatalf("want one warning naming the path, got %+v", evs)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// buildSmokeTree recreates the README smoke-test filesystem layout
|
// buildSmokeTree recreates the README smoke-test filesystem layout
|
||||||
// with deterministic content and returns the tree root.
|
// with deterministic content and returns the tree root.
|
||||||
func buildSmokeTree(t *testing.T) string {
|
func buildSmokeTree(t *testing.T) string {
|
||||||
@@ -1215,7 +1058,7 @@ func TestScanPipeline(t *testing.T) {
|
|||||||
db := openTestDB(t)
|
db := openTestDB(t)
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: smokeTreeFiles, added: smokeTreeFiles}) {
|
if st != (scanStats{added: smokeTreeFiles}) {
|
||||||
t.Fatalf("stats = %+v, want %d added only", st, smokeTreeFiles)
|
t.Fatalf("stats = %+v, want %d added only", st, smokeTreeFiles)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1238,7 +1081,7 @@ func TestSyncScanUnchangedReuse(t *testing.T) {
|
|||||||
writeFile(t, dir, "b.bin", pattern(2, 600))
|
writeFile(t, dir, "b.bin", pattern(2, 600))
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 2, added: 2}) {
|
if st != (scanStats{added: 2}) {
|
||||||
t.Fatalf("first scan stats = %+v, want 2 added", st)
|
t.Fatalf("first scan stats = %+v, want 2 added", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1252,7 +1095,7 @@ func TestSyncScanUnchangedReuse(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
st = syncTree(t, db, dir)
|
st = syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 2, unchanged: 2}) {
|
if st != (scanStats{unchanged: 2}) {
|
||||||
t.Fatalf("rescan stats = %+v, want 2 unchanged", st)
|
t.Fatalf("rescan stats = %+v, want 2 unchanged", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1281,7 +1124,7 @@ func TestSyncScanMtimeBump(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 1, updated: 1}) {
|
if st != (scanStats{updated: 1}) {
|
||||||
t.Fatalf("mtime-bump stats = %+v, want 1 updated", st)
|
t.Fatalf("mtime-bump stats = %+v, want 1 updated", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1309,7 +1152,7 @@ func TestSyncScanAddRemove(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 2, added: 1, removed: 1, unchanged: 1}) {
|
if st != (scanStats{added: 1, removed: 1, unchanged: 1}) {
|
||||||
t.Fatalf("add/remove stats = %+v, want 1 added 1 removed 1 unchanged",
|
t.Fatalf("add/remove stats = %+v, want 1 added 1 removed 1 unchanged",
|
||||||
st)
|
st)
|
||||||
}
|
}
|
||||||
@@ -1426,7 +1269,7 @@ func TestSyncScanOverlappingRoots(t *testing.T) {
|
|||||||
// A file reachable via two overlapping operands is deduplicated
|
// A file reachable via two overlapping operands is deduplicated
|
||||||
// by path in the shared walk and processed once.
|
// by path in the shared walk and processed once.
|
||||||
st := syncTree(t, db, dir, filepath.Join(dir, "sub"))
|
st := syncTree(t, db, dir, filepath.Join(dir, "sub"))
|
||||||
if st != (scanStats{walked: 1, added: 1}) {
|
if st != (scanStats{added: 1}) {
|
||||||
t.Fatalf("stats = %+v, want 1 added", st)
|
t.Fatalf("stats = %+v, want 1 added", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1447,7 +1290,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
|||||||
// Neither size is shared, so neither file is read: both records
|
// Neither size is shared, so neither file is read: both records
|
||||||
// are written without hashes and no duplicates are reported.
|
// are written without hashes and no duplicates are reported.
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 2, added: 2}) {
|
if st != (scanStats{added: 2}) {
|
||||||
t.Fatalf("stats = %+v, want 2 added", st)
|
t.Fatalf("stats = %+v, want 2 added", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1469,7 +1312,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
|||||||
c := writeFile(t, dir, "c.bin", pattern(1, 500))
|
c := writeFile(t, dir, "c.bin", pattern(1, 500))
|
||||||
|
|
||||||
st = syncTree(t, db, dir)
|
st = syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 3, added: 1, updated: 1, unchanged: 1}) {
|
if st != (scanStats{added: 1, updated: 1, unchanged: 1}) {
|
||||||
t.Fatalf("rescan stats = %+v, want 1 added 1 updated 1 unchanged",
|
t.Fatalf("rescan stats = %+v, want 1 added 1 updated 1 unchanged",
|
||||||
st)
|
st)
|
||||||
}
|
}
|
||||||
@@ -1533,7 +1376,7 @@ func TestScanHardlinksReadOnce(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
st := syncTree(t, db, dir)
|
st := syncTree(t, db, dir)
|
||||||
if st != (scanStats{walked: 2, added: 2}) {
|
if st != (scanStats{added: 2}) {
|
||||||
t.Fatalf("stats = %+v, want 2 added", st)
|
t.Fatalf("stats = %+v, want 2 added", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1655,13 +1498,6 @@ func injectWriteFailure(t *testing.T, path string) {
|
|||||||
func baselineGoroutines(t *testing.T) int {
|
func baselineGoroutines(t *testing.T) int {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
// The first scan command in a process starts os/signal's goroutine,
|
|
||||||
// which never exits. Start it now, so the baseline counts it
|
|
||||||
// instead of the scan seeming to leave it behind.
|
|
||||||
ch := make(chan os.Signal, 1)
|
|
||||||
signal.Notify(ch, syscall.SIGINT)
|
|
||||||
signal.Stop(ch)
|
|
||||||
|
|
||||||
deadline := time.Now().Add(goroutineSettle)
|
deadline := time.Now().Add(goroutineSettle)
|
||||||
last := runtime.NumGoroutine()
|
last := runtime.NumGoroutine()
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user