Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c2804d088b |
+1
-6
@@ -1,9 +1,4 @@
|
|||||||
# .git is sent without its config. Without a VERSION build argument the
|
.git
|
||||||
# stage that compiles runs `git describe --tags --always` on .git, which
|
|
||||||
# does not need .git/config; that file can hold a credential, such as a
|
|
||||||
# password in a remote URL or the token the CI checkout step stores there.
|
|
||||||
.git/config
|
|
||||||
|
|
||||||
.claude
|
.claude
|
||||||
.DS_Store
|
.DS_Store
|
||||||
sfdupes
|
sfdupes
|
||||||
|
|||||||
@@ -6,6 +6,4 @@ jobs:
|
|||||||
steps:
|
steps:
|
||||||
# actions/checkout v4.2.2, 2026-02-22
|
# actions/checkout v4.2.2, 2026-02-22
|
||||||
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
|
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
|
||||||
with:
|
|
||||||
fetch-depth: 0
|
|
||||||
- run: script/cibuild
|
- run: script/cibuild
|
||||||
|
|||||||
+1
-13
@@ -108,19 +108,7 @@ ARG CHECK_EPOCH
|
|||||||
RUN echo "gate test, epoch ${CHECK_EPOCH}" && make test
|
RUN echo "gate test, epoch ${CHECK_EPOCH}" && make test
|
||||||
RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
|
RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check
|
||||||
|
|
||||||
# The version stamped into the binary: the VERSION build argument when
|
RUN make build
|
||||||
# one is given, otherwise `git describe --tags --always` of the .git in
|
|
||||||
# the build context (git is installed by script/bootstrap above). A
|
|
||||||
# context that carries .git and still yields no version fails the build;
|
|
||||||
# with neither, as from a source tarball, it is "dev".
|
|
||||||
ARG VERSION
|
|
||||||
RUN version="${VERSION:-$(git describe --tags --always || echo dev)}"; \
|
|
||||||
if [ -e .git ] && { [ -z "$version" ] || [ "$version" = dev ] || \
|
|
||||||
[ "$version" = unknown ]; }; then \
|
|
||||||
echo "no version could be derived although the build context carries .git" >&2; \
|
|
||||||
exit 1; \
|
|
||||||
fi; \
|
|
||||||
make build VERSION="$version"
|
|
||||||
|
|
||||||
# Runtime stage
|
# Runtime stage
|
||||||
# alpine:3.22, 2026-07-23
|
# alpine:3.22, 2026-07-23
|
||||||
|
|||||||
@@ -94,15 +94,7 @@ Goals, in order:
|
|||||||
millions of files, ~150 TB filesystem, possibly slow or busy disks
|
millions of files, ~150 TB filesystem, possibly slow or busy disks
|
||||||
(ZFS pool under resilver). Holding one small record (path, size,
|
(ZFS pool under resilver). Holding one small record (path, size,
|
||||||
mtime) per file in memory during a scan is acceptable; holding
|
mtime) per file in memory during a scan is acceptable; holding
|
||||||
every file's hashes is not (they stay in the database). The
|
every file's hashes is not (they stay in the database).
|
||||||
reporting commands do not hold every file's hashes either: `report`
|
|
||||||
lets SQLite group and order the records and writes each row as it
|
|
||||||
reads it, so its memory does not grow with the database, and
|
|
||||||
`trees` reads the records in path order and keeps each directory's
|
|
||||||
path, digest and totals, plus the hashes of only the files in the
|
|
||||||
directories holding the record being read, so its memory grows with
|
|
||||||
the number of directories and with the size of the largest
|
|
||||||
directory.
|
|
||||||
3. **Scan incrementally, analyze offline.** The expensive filesystem
|
3. **Scan incrementally, analyze offline.** The expensive filesystem
|
||||||
scan maintains a persistent database; an unchanged file is never
|
scan maintains a persistent database; an unchanged file is never
|
||||||
read again on a rescan, except to compute its content hash once a
|
read again on a rescan, except to compute its content hash once a
|
||||||
@@ -121,10 +113,8 @@ Goals, in order:
|
|||||||
`sfdupes`.
|
`sfdupes`.
|
||||||
- Dependencies: standard library, `github.com/spf13/cobra` for the
|
- Dependencies: standard library, `github.com/spf13/cobra` for the
|
||||||
CLI, **one progress-bar library**
|
CLI, **one progress-bar library**
|
||||||
(`github.com/schollz/progressbar/v3`), `golang.org/x/term` to tell
|
(`github.com/schollz/progressbar/v3`), and **one SQLite driver**
|
||||||
whether stderr is a terminal, **one SQLite driver**
|
(`modernc.org/sqlite`, pure Go, so builds keep cgo disabled).
|
||||||
(`modernc.org/sqlite`, pure Go, so builds keep cgo disabled), and
|
|
||||||
`golang.org/x/sys` for `flock(2)` (the scan lock, see "Database").
|
|
||||||
`github.com/spf13/viper` is permitted if configuration-file support
|
`github.com/spf13/viper` is permitted if configuration-file support
|
||||||
is ever needed, but is not currently used. No other third-party
|
is ever needed, but is not currently used. No other third-party
|
||||||
deps.
|
deps.
|
||||||
@@ -162,45 +152,16 @@ All three subcommands operate on a single SQLite database file:
|
|||||||
use. `report` and `trees` require an existing database; a missing
|
use. `report` and `trees` require an existing database; a missing
|
||||||
database file is a fatal error (exit 1) telling the user to run
|
database file is a fatal error (exit 1) telling the user to run
|
||||||
`scan` first.
|
`scan` first.
|
||||||
- Only one `scan` runs against a database at a time. For its whole
|
- The database uses WAL journal mode and a busy timeout, so running a
|
||||||
run, `scan` holds an exclusive `flock(2)` lock on a lock file
|
report while a cron `scan` is in progress is safe. The filesystem
|
||||||
beside the database, named by appending `.lock` to the database
|
is authoritative; the database is an eventually-consistent
|
||||||
path (`/var/lib/sfdupes/db.sqlite.lock` by default), taken before
|
reflection of it. Hashed records are committed in batched
|
||||||
it walks the filesystem or opens the database. A second `scan`
|
transactions while the scan is still running (keeping the WAL
|
||||||
against the same database does not wait: it fails at once with a
|
small and letting concurrent reports observe progress), so a
|
||||||
one-line error naming the lock file and exits 1, without walking
|
report may see a scan's changes partially applied, and a scan
|
||||||
anything or opening the database, and the running scan carries on.
|
that dies partway leaves a valid database holding everything
|
||||||
The lock file is created on first use, open to its owner only, and
|
hashed so far; the next scan skips those records and converges
|
||||||
left in place: a leftover file blocks nothing, because the lock
|
toward the filesystem.
|
||||||
ends with the process holding it however it ends, a fatal error or
|
|
||||||
an interrupt included, and deleting the file while a scan runs
|
|
||||||
would let a second scan start. `report` and `trees` never take the
|
|
||||||
lock, so they run during a scan.
|
|
||||||
- While `scan` runs, the database is in WAL journal mode with a busy
|
|
||||||
timeout, so running a report while a cron `scan` is in progress is
|
|
||||||
safe. The filesystem is authoritative; the database is an
|
|
||||||
eventually-consistent reflection of it. Hashed records are
|
|
||||||
committed in batched transactions while the scan is still running
|
|
||||||
(keeping the WAL small and letting concurrent reports observe
|
|
||||||
progress), so a report may see a scan's changes partially applied,
|
|
||||||
and a scan that dies partway leaves a valid database holding
|
|
||||||
everything hashed so far; the next scan skips those records and
|
|
||||||
converges toward the filesystem.
|
|
||||||
- `scan` switches the database back to rollback-journal mode when it
|
|
||||||
closes it, so between scans the database file alone holds the whole
|
|
||||||
database. Each switch needs the database to itself: a `scan` that
|
|
||||||
starts while a report is still reading waits for it up to the
|
|
||||||
10-second busy timeout, then fails; a `scan` that ends while a
|
|
||||||
report has the database open warns and leaves the database in WAL
|
|
||||||
mode until the next scan. `report` writes each row as it reads it,
|
|
||||||
so it is still reading while its output is paused (a pager, a
|
|
||||||
stalled pipe), and a `scan` started then fails after the busy
|
|
||||||
timeout.
|
|
||||||
- `report` and `trees` open the database read-only and need only read
|
|
||||||
access to the database file, and no write access to its directory.
|
|
||||||
While the database is in WAL mode they also read the `-wal` and
|
|
||||||
`-shm` files beside it, which SQLite creates with the database
|
|
||||||
file's permissions.
|
|
||||||
- Schema (`PRAGMA user_version` is the schema version, currently 1; a
|
- Schema (`PRAGMA user_version` is the schema version, currently 1; a
|
||||||
database with any other version is a fatal error):
|
database with any other version is a fatal error):
|
||||||
|
|
||||||
@@ -213,7 +174,6 @@ All three subcommands operate on a single SQLite database file:
|
|||||||
tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB
|
tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB
|
||||||
content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples
|
content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples
|
||||||
) WITHOUT ROWID;
|
) WITHOUT ROWID;
|
||||||
CREATE INDEX files_signature ON files (size, head, tail, content);
|
|
||||||
```
|
```
|
||||||
|
|
||||||
Paths are stored as BLOBs because Unix paths are raw bytes, not
|
Paths are stored as BLOBs because Unix paths are raw bytes, not
|
||||||
@@ -228,8 +188,7 @@ All three subcommands operate on a single SQLite database file:
|
|||||||
until the content phase of a scan (see "`scan` mode" below) has
|
until the content phase of a scan (see "`scan` mode" below) has
|
||||||
read the file. A record with an empty `content` is never part of a
|
read the file. A record with an empty `content` is never part of a
|
||||||
duplicate group, though it still defines the file for tree
|
duplicate group, though it still defines the file for tree
|
||||||
reconstruction. The `files_signature` index lets SQLite group the
|
reconstruction.
|
||||||
records by signature for `report` without sorting the whole table.
|
|
||||||
|
|
||||||
### Duplicate detection
|
### Duplicate detection
|
||||||
|
|
||||||
@@ -292,18 +251,6 @@ duplicates another or lies under another is dropped before walking,
|
|||||||
so every file is reached exactly once and produces one database
|
so every file is reached exactly once and produces one database
|
||||||
record.
|
record.
|
||||||
|
|
||||||
An operand that is a symlink (never followed, not even as an operand),
|
|
||||||
socket, FIFO, or device node, or a directory named `.zfs`, is not
|
|
||||||
scanned. `scan` prints a one-line warning naming the path and what it
|
|
||||||
is, counts it as skipped, and drops it from the scanned operands before
|
|
||||||
reading the database. Another operand beneath it is still scanned. The
|
|
||||||
records stored beneath it are not deleted: they are treated like any
|
|
||||||
other record outside the scanned operands, including the content-phase
|
|
||||||
exception below. If it lies under another operand, they are under that
|
|
||||||
operand instead, and are deleted like any other record there that this
|
|
||||||
scan did not verify. This is not an error: a scan whose every operand
|
|
||||||
is dropped walks nothing and exits 0.
|
|
||||||
|
|
||||||
`scan` synchronizes the database with the filesystem state under the
|
`scan` synchronizes the database with the filesystem state under the
|
||||||
scanned operands:
|
scanned operands:
|
||||||
|
|
||||||
@@ -409,12 +356,9 @@ during the hash phase:
|
|||||||
Rules for the walk:
|
Rules for the walk:
|
||||||
|
|
||||||
- Only regular files. Skip directories, symlinks (do not follow,
|
- Only regular files. Skip directories, symlinks (do not follow,
|
||||||
including symlink operands), sockets, FIFOs, and device nodes. An
|
including symlink operands), sockets, FIFOs, and device nodes.
|
||||||
operand that is a symlink, socket, FIFO, or device node is dropped
|
|
||||||
as described in "`scan` mode" above.
|
|
||||||
- Never descend into a directory named `.zfs` (ZFS snapshot pseudo-dirs;
|
- Never descend into a directory named `.zfs` (ZFS snapshot pseudo-dirs;
|
||||||
walking them would list every file once per snapshot), not even
|
walking them would list every file once per snapshot).
|
||||||
when it is an operand; such an operand is dropped the same way.
|
|
||||||
- Filesystem boundaries are crossed by default. With `-x`
|
- Filesystem boundaries are crossed by default. With `-x`
|
||||||
(long form `--one-file-system`, following the GNU `du`/`rsync`
|
(long form `--one-file-system`, following the GNU `du`/`rsync`
|
||||||
convention), never descend into a directory on a different
|
convention), never descend into a directory on a different
|
||||||
@@ -425,11 +369,9 @@ Rules for the walk:
|
|||||||
path, and continue. Per-file errors never abort the run; the final
|
path, and continue. Per-file errors never abort the run; the final
|
||||||
summary reports how many were skipped. As specified above, a
|
summary reports how many were skipped. As specified above, a
|
||||||
skipped path that has a database record from an earlier scan loses
|
skipped path that has a database record from an earlier scan loses
|
||||||
that record, unless it failed only in the content phase, or is an
|
that record, unless it failed only in the content phase; an
|
||||||
operand dropped before the database was read that lies under no
|
unreadable directory subtree likewise loses its records (accepted:
|
||||||
other operand; an unreadable directory subtree likewise loses its
|
the database mirrors what the latest scan could actually verify).
|
||||||
records (accepted: the database mirrors what the latest scan could
|
|
||||||
actually verify).
|
|
||||||
|
|
||||||
Concurrency: the walk phase (which also stats files), the hash phase,
|
Concurrency: the walk phase (which also stats files), the hash phase,
|
||||||
and the content phase each use a worker pool of `--workers` workers
|
and the content phase each use a worker pool of `--workers` workers
|
||||||
@@ -457,12 +399,9 @@ arguments.
|
|||||||
|
|
||||||
**`report` must never touch the filesystem being analyzed.** It does not
|
**`report` must never touch the filesystem being analyzed.** It does not
|
||||||
stat, open, or otherwise access any path that appears in the records; its
|
stat, open, or otherwise access any path that appears in the records; its
|
||||||
only I/O is reading the database, writing stdout/stderr, and the
|
only I/O is reading the database and writing stdout/stderr. It must
|
||||||
temporary file SQLite sorts in when the duplicate rows do not fit in
|
produce identical output whether or not the scanned filesystem is still
|
||||||
memory. SQLite puts that file in `$SQLITE_TMPDIR` or `$TMPDIR` when set,
|
mounted.
|
||||||
otherwise in `/var/tmp` (or `/tmp`), and deletes it as soon as it has
|
|
||||||
opened it. `report` must produce identical output whether or not the
|
|
||||||
scanned filesystem is still mounted.
|
|
||||||
|
|
||||||
Processing:
|
Processing:
|
||||||
|
|
||||||
@@ -489,15 +428,6 @@ first dupe size
|
|||||||
/srv/a/big.iso /srv/c/big-copy2.iso 4294967296
|
/srv/a/big.iso /srv/c/big-copy2.iso 4294967296
|
||||||
```
|
```
|
||||||
|
|
||||||
Paths are raw bytes and may hold any byte except NUL, so the path
|
|
||||||
columns (`first` and `dupe`) are escaped to keep every row one line of
|
|
||||||
tab-separated fields: a backslash is written as `\\`, a tab as `\t`, a
|
|
||||||
newline as `\n`, and a carriage return as `\r`. Every other byte is
|
|
||||||
written unchanged, including bytes that are not valid UTF-8. Undoing
|
|
||||||
those four escapes gives back the stored path. Grouping and ordering
|
|
||||||
use the stored path, not the escaped one. The warnings `scan` prints on
|
|
||||||
stderr are escaped the same way, so each warning is one line.
|
|
||||||
|
|
||||||
Summary to stderr: records read, number of duplicate groups, number of
|
Summary to stderr: records read, number of duplicate groups, number of
|
||||||
dupe files, and total reclaimable bytes (sum of `size` over all dupe
|
dupe files, and total reclaimable bytes (sum of `size` over all dupe
|
||||||
rows) in human units.
|
rows) in human units.
|
||||||
@@ -569,9 +499,6 @@ first dupe files size
|
|||||||
/srv/a/project /srv/backup/project 3417 104857600
|
/srv/a/project /srv/backup/project 3417 104857600
|
||||||
```
|
```
|
||||||
|
|
||||||
The `first` and `dupe` paths are escaped as described under "Report
|
|
||||||
output format". The root directory's path is `/`.
|
|
||||||
|
|
||||||
Summary to stderr: records read, number of duplicate-tree groups,
|
Summary to stderr: records read, number of duplicate-tree groups,
|
||||||
number of dupe trees, and total reclaimable bytes (sum of `size` over
|
number of dupe trees, and total reclaimable bytes (sum of `size` over
|
||||||
all dupe rows) in human units.
|
all dupe rows) in human units.
|
||||||
@@ -606,41 +533,22 @@ hash: [12345/98765] 12% |████ | 92 files/s elapsed 2:32 eta 17:54
|
|||||||
|
|
||||||
Additional requirements:
|
Additional requirements:
|
||||||
|
|
||||||
- When stderr is not a terminal (a pipe, a file, `/dev/null`), do not
|
- When stderr is not a TTY, do not emit ANSI redraws: print a plain
|
||||||
emit ANSI redraws: print a plain one-line progress update the moment
|
one-line progress update no more often than every 5 seconds instead.
|
||||||
each phase starts, then no more often than every 5 seconds.
|
|
||||||
- Progress updates are driven from the main goroutine and must be
|
- Progress updates are driven from the main goroutine and must be
|
||||||
non-blocking with respect to the worker pool. On a terminal the
|
non-blocking with respect to the worker pool.
|
||||||
spinner-style displays also redraw on their own several times a
|
|
||||||
second, so their count and elapsed time stay current while a phase
|
|
||||||
waits for its next item.
|
|
||||||
- A warning printed during a phase always lands on a line of its own,
|
|
||||||
never inside the progress display.
|
|
||||||
- `report` and `trees` modes need no progress display, only their
|
- `report` and `trees` modes need no progress display, only their
|
||||||
stderr summaries.
|
stderr summaries.
|
||||||
|
|
||||||
### Error handling and exit codes
|
### Error handling and exit codes
|
||||||
|
|
||||||
- `0`: success, even if individual files were skipped with warnings.
|
- `0`: success, even if individual files were skipped with warnings.
|
||||||
- `1`: fatal error (e.g., a `PATH` operand does not exist, another
|
- `1`: fatal error (e.g., a `PATH` operand does not exist, the
|
||||||
`scan` is already running against the same database, the database
|
database cannot be created/opened/read/written, a missing database
|
||||||
cannot be created/opened/read/written, a missing database for
|
for `report`/`trees`, stdout write failure).
|
||||||
`report`/`trees`, stdout write failure).
|
|
||||||
- `2`: usage error (including `scan` with no `PATH` operand and
|
- `2`: usage error (including `scan` with no `PATH` operand and
|
||||||
`report`/`trees` with any positional argument).
|
`report`/`trees` with any positional argument).
|
||||||
|
|
||||||
A stdout write failure, such as a full disk, is reported in one line on
|
|
||||||
stderr and exits 1. Two cases never reach sfdupes as a failed write:
|
|
||||||
|
|
||||||
- When the reader of a stdout pipe exits early, as in
|
|
||||||
`sfdupes report | head`, the next write ends sfdupes with `SIGPIPE`,
|
|
||||||
quietly and without a summary, the way `cat` or `sort` end. The
|
|
||||||
shell reports the signal (status 141 in most shells), not exit 1.
|
|
||||||
- When stdout is closed outright (`sfdupes report >&-`), the Go
|
|
||||||
runtime opens `/dev/null` in its place before sfdupes starts, so
|
|
||||||
the output is discarded and the run succeeds, as with
|
|
||||||
`> /dev/null`.
|
|
||||||
|
|
||||||
## Entrypoints
|
## Entrypoints
|
||||||
|
|
||||||
This repository adheres to the
|
This repository adheres to the
|
||||||
@@ -778,7 +686,7 @@ All of the following, run in this directory, must pass:
|
|||||||
|
|
||||||
```sh
|
```sh
|
||||||
d=$(mktemp -d)
|
d=$(mktemp -d)
|
||||||
export SFDUPES_DATABASE="$(mktemp -d)/db.sqlite"
|
export SFDUPES_DATABASE="$d/db.sqlite"
|
||||||
mkdir -p "$d/a" "$d/b"
|
mkdir -p "$d/a" "$d/b"
|
||||||
head -c 2000 /dev/urandom > "$d/a/one.bin"
|
head -c 2000 /dev/urandom > "$d/a/one.bin"
|
||||||
cp "$d/a/one.bin" "$d/b/copy.bin"
|
cp "$d/a/one.bin" "$d/b/copy.bin"
|
||||||
@@ -806,9 +714,9 @@ All of the following, run in this directory, must pass:
|
|||||||
./sfdupes report
|
./sfdupes report
|
||||||
```
|
```
|
||||||
|
|
||||||
(The database lives in a temp directory of its own: inside `$d`,
|
(The scan database lives inside `$d` here purely for test hygiene;
|
||||||
the scan would record it, and its empty lock file would join the
|
scanning `$d` therefore also records the SQLite file itself, which
|
||||||
`empty1`/`empty2` group.)
|
is harmless.)
|
||||||
|
|
||||||
Expected from the first `report`: `one.bin`/`copy.bin`/`copy2.bin`
|
Expected from the first `report`: `one.bin`/`copy.bin`/`copy2.bin`
|
||||||
form one group (two dupe rows, `first` is the lexicographically
|
form one group (two dupe rows, `first` is the lexicographically
|
||||||
|
|||||||
@@ -29,45 +29,6 @@
|
|||||||
|
|
||||||
# Completed Steps
|
# Completed Steps
|
||||||
|
|
||||||
- `report` and `trees` stream the records instead of holding them all in
|
|
||||||
memory; the schema gains the `files_signature` index (2026-10-04,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/14)
|
|
||||||
|
|
||||||
- progress prints at once on a non-terminal, uses a real terminal test,
|
|
||||||
and prints warnings through a spinner instead of racing its redraw
|
|
||||||
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/13)
|
|
||||||
|
|
||||||
- warn about and skip symlink, socket, FIFO, device and `.zfs`
|
|
||||||
operands, keeping the records beneath them (2026-10-03,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/9)
|
|
||||||
|
|
||||||
- `scan` holds a lock on a lock file beside the database for its whole run,
|
|
||||||
so a second `scan` fails at once with exit 1 (2026-10-03,
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/53)
|
|
||||||
|
|
||||||
- test stdout write failures in `report` and `trees`; README states that
|
|
||||||
`| head` ends sfdupes by `SIGPIPE` and `>&-` writes to `/dev/null`
|
|
||||||
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/30)
|
|
||||||
|
|
||||||
- `report` and `trees` open the database read-only, and `scan` leaves it
|
|
||||||
out of WAL mode, so reading needs only read access (2026-10-03, closes
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/8)
|
|
||||||
|
|
||||||
- escape tabs, newlines, carriage returns and backslashes in report,
|
|
||||||
trees and warning paths; the root directory's path is `/`
|
|
||||||
(2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/7)
|
|
||||||
|
|
||||||
- stamp the git tag or short commit in a plain `docker build .`
|
|
||||||
instead of `dev` (2026-10-02, branch `next`, closes
|
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now
|
|
||||||
sends `.git`, without `.git/config`, and the `Dockerfile` build
|
|
||||||
stage takes the `VERSION` build argument when one is given,
|
|
||||||
otherwise `git describe --tags --always` of that `.git`. The build
|
|
||||||
fails if the context carries `.git` and the version still comes out
|
|
||||||
empty, `dev` or `unknown`. The CI checkout step fetches the full
|
|
||||||
history (`fetch-depth: 0`) so CI sees the tag and stamps the same
|
|
||||||
value as `make build`.
|
|
||||||
|
|
||||||
- replace the 1 KiB end-window sampling with the head/tail plus
|
- replace the 1 KiB end-window sampling with the head/tail plus
|
||||||
content-hash ladder (2026-09-22, branch `next`, closes
|
content-hash ladder (2026-09-22, branch `next`, closes
|
||||||
https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is
|
https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is
|
||||||
|
|||||||
@@ -11,7 +11,6 @@ import (
|
|||||||
"slices"
|
"slices"
|
||||||
"strconv"
|
"strconv"
|
||||||
|
|
||||||
"golang.org/x/sys/unix"
|
|
||||||
// The pure-Go SQLite driver, registered as "sqlite"; keeps cgo
|
// The pure-Go SQLite driver, registered as "sqlite"; keeps cgo
|
||||||
// disabled.
|
// disabled.
|
||||||
_ "modernc.org/sqlite"
|
_ "modernc.org/sqlite"
|
||||||
@@ -33,11 +32,6 @@ const schemaVersion = 1
|
|||||||
// scan.
|
// scan.
|
||||||
const dbDirPerm = 0o755
|
const dbDirPerm = 0o755
|
||||||
|
|
||||||
// lockFilePerm is the mode for the scan lock file. Anyone who can open
|
|
||||||
// the file can hold the lock and keep every scan from running, so it
|
|
||||||
// is open to its owner only.
|
|
||||||
const lockFilePerm = 0o600
|
|
||||||
|
|
||||||
// createTableSQL is the schema applied to a fresh database. Paths are
|
// createTableSQL is the schema applied to a fresh database. Paths are
|
||||||
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8.
|
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8.
|
||||||
const createTableSQL = `
|
const createTableSQL = `
|
||||||
@@ -51,12 +45,6 @@ CREATE TABLE files (
|
|||||||
) WITHOUT ROWID
|
) WITHOUT ROWID
|
||||||
`
|
`
|
||||||
|
|
||||||
// createIndexSQL indexes the records by signature, so report can have
|
|
||||||
// SQLite group them without sorting the whole table.
|
|
||||||
const createIndexSQL = `
|
|
||||||
CREATE INDEX files_signature ON files (size, head, tail, content)
|
|
||||||
`
|
|
||||||
|
|
||||||
// upsertSQL inserts one file record, replacing any existing record for
|
// upsertSQL inserts one file record, replacing any existing record for
|
||||||
// the same path.
|
// the same path.
|
||||||
const upsertSQL = `
|
const upsertSQL = `
|
||||||
@@ -76,10 +64,6 @@ var errNoDatabase = errors.New(
|
|||||||
// does not understand.
|
// does not understand.
|
||||||
var errSchemaVersion = errors.New("unsupported database schema version")
|
var errSchemaVersion = errors.New("unsupported database schema version")
|
||||||
|
|
||||||
// errScanRunning reports that another scan holds the lock on the
|
|
||||||
// database.
|
|
||||||
var errScanRunning = errors.New("another scan is running")
|
|
||||||
|
|
||||||
// databasePath resolves the database location: SFDUPES_DATABASE when
|
// databasePath resolves the database location: SFDUPES_DATABASE when
|
||||||
// set and non-empty, the compiled-in default otherwise.
|
// set and non-empty, the compiled-in default otherwise.
|
||||||
func databasePath() string {
|
func databasePath() string {
|
||||||
@@ -90,24 +74,16 @@ func databasePath() string {
|
|||||||
return defaultDatabasePath
|
return defaultDatabasePath
|
||||||
}
|
}
|
||||||
|
|
||||||
// scanParams are the connection parameters for scan: read-write, with
|
// openDB opens the SQLite database at path with WAL journaling and a
|
||||||
// WAL journaling and a busy timeout, so a report can run while a cron
|
// busy timeout, so a report can run while a cron scan is in progress.
|
||||||
// scan is in progress. closeScanDatabase leaves WAL mode again.
|
// It does not create or verify the schema.
|
||||||
const scanParams = "_pragma=busy_timeout(10000)" +
|
func openDB(path string) (*sql.DB, error) {
|
||||||
"&_pragma=journal_mode(WAL)" +
|
dsn := "file:" + path +
|
||||||
"&_pragma=synchronous(NORMAL)"
|
"?_pragma=busy_timeout(10000)" +
|
||||||
|
"&_pragma=journal_mode(WAL)" +
|
||||||
|
"&_pragma=synchronous(NORMAL)"
|
||||||
|
|
||||||
// reportParams are the connection parameters for report and trees:
|
db, err := sql.Open("sqlite", dsn)
|
||||||
// read-only, with the same busy timeout. They set no journal mode,
|
|
||||||
// because setting one is a write.
|
|
||||||
const reportParams = "mode=ro" +
|
|
||||||
"&_pragma=busy_timeout(10000)" +
|
|
||||||
"&_pragma=query_only(1)"
|
|
||||||
|
|
||||||
// openDB opens the SQLite database at path with the connection
|
|
||||||
// parameters params. It does not create or verify the schema.
|
|
||||||
func openDB(path, params string) (*sql.DB, error) {
|
|
||||||
db, err := sql.Open("sqlite", "file:"+path+"?"+params)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("open database %s: %w", path, err)
|
return nil, fmt.Errorf("open database %s: %w", path, err)
|
||||||
}
|
}
|
||||||
@@ -120,43 +96,6 @@ func openDB(path, params string) (*sql.DB, error) {
|
|||||||
return db, nil
|
return db, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// lockScanDatabase takes the lock that keeps a second scan off the
|
|
||||||
// database at path: an exclusive flock(2) on the file beside it named
|
|
||||||
// path with ".lock" appended, created along with the database's parent
|
|
||||||
// directory if missing. A lock held by another scan fails at once
|
|
||||||
// instead of waiting. The lock lasts until the returned file is closed
|
|
||||||
// or the process ends. The file is never deleted: a scan that deleted
|
|
||||||
// it would let the next scan lock a new file while another still holds
|
|
||||||
// the old one.
|
|
||||||
func lockScanDatabase(path string) (*os.File, error) {
|
|
||||||
err := os.MkdirAll(filepath.Dir(path), dbDirPerm)
|
|
||||||
if err != nil {
|
|
||||||
return nil, fmt.Errorf("create database directory: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
lockPath := path + ".lock"
|
|
||||||
|
|
||||||
//nolint:gosec // the operator chooses the database path
|
|
||||||
f, err := os.OpenFile(lockPath, os.O_RDWR|os.O_CREATE, lockFilePerm)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
|
|
||||||
err = unix.Flock(int(f.Fd()), unix.LOCK_EX|unix.LOCK_NB)
|
|
||||||
if err != nil {
|
|
||||||
_ = f.Close()
|
|
||||||
|
|
||||||
if errors.Is(err, unix.EWOULDBLOCK) {
|
|
||||||
return nil, fmt.Errorf("%w (lock held on %s)",
|
|
||||||
errScanRunning, lockPath)
|
|
||||||
}
|
|
||||||
|
|
||||||
return nil, fmt.Errorf("lock %s: %w", lockPath, err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return f, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// openScanDatabase opens the database for the scan subcommand, creating
|
// openScanDatabase opens the database for the scan subcommand, creating
|
||||||
// the file, its parent directory, and the schema as needed.
|
// the file, its parent directory, and the schema as needed.
|
||||||
func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
|
func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
|
||||||
@@ -165,7 +104,7 @@ func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
|
|||||||
return nil, fmt.Errorf("create database directory: %w", err)
|
return nil, fmt.Errorf("create database directory: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
db, err := openDB(path, scanParams)
|
db, err := openDB(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
@@ -180,24 +119,6 @@ func openScanDatabase(ctx context.Context, path string) (*sql.DB, error) {
|
|||||||
return db, nil
|
return db, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// closeScanDatabase switches the database at path from WAL back to
|
|
||||||
// rollback-journal mode and closes it. Out of WAL mode the database
|
|
||||||
// file alone holds the whole database, so a reader needs no -wal or
|
|
||||||
// -shm file beside it, nor write access to create them. The switch
|
|
||||||
// fails while a report has the database open; the database then stays
|
|
||||||
// in WAL mode, still readable, until a later scan closes it.
|
|
||||||
func closeScanDatabase(ctx context.Context, db *sql.DB, path string) {
|
|
||||||
// Runs on the way out of a cancelled scan too.
|
|
||||||
_, err := db.ExecContext(context.WithoutCancel(ctx),
|
|
||||||
"PRAGMA journal_mode = DELETE")
|
|
||||||
if err != nil {
|
|
||||||
fmt.Fprintf(os.Stderr, "scan: database %s left in WAL mode: %v\n",
|
|
||||||
path, err)
|
|
||||||
}
|
|
||||||
|
|
||||||
_ = db.Close()
|
|
||||||
}
|
|
||||||
|
|
||||||
// openReportDatabase opens an existing database for the report and
|
// openReportDatabase opens an existing database for the report and
|
||||||
// trees subcommands. A missing database file is an error directing the
|
// trees subcommands. A missing database file is an error directing the
|
||||||
// user to run scan first; the schema version must match exactly.
|
// user to run scan first; the schema version must match exactly.
|
||||||
@@ -213,7 +134,7 @@ func openReportDatabase(ctx context.Context,
|
|||||||
return nil, fmt.Errorf("database: %w", err)
|
return nil, fmt.Errorf("database: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
db, err := openDB(path, reportParams)
|
db, err := openDB(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
@@ -262,11 +183,6 @@ func createSchema(ctx context.Context, db *sql.DB) error {
|
|||||||
return fmt.Errorf("create schema: %w", err)
|
return fmt.Errorf("create schema: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
_, err = db.ExecContext(ctx, createIndexSQL)
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("create schema: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err = db.ExecContext(ctx,
|
_, err = db.ExecContext(ctx,
|
||||||
"PRAGMA user_version = "+strconv.Itoa(schemaVersion))
|
"PRAGMA user_version = "+strconv.Itoa(schemaVersion))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -288,18 +204,18 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) {
|
|||||||
return v, nil
|
return v, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// loadFileRows streams every record to fn in path order: byte order,
|
// loadFileRows reads every record from the files table.
|
||||||
// which is the order of the primary key, so SQLite does not sort.
|
func loadFileRows(ctx context.Context, db *sql.DB) ([]scanRec, error) {
|
||||||
func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
|
|
||||||
rows, err := db.QueryContext(ctx,
|
rows, err := db.QueryContext(ctx,
|
||||||
"SELECT path, size, mtime, head, tail, content FROM files "+
|
"SELECT path, size, mtime, head, tail, content FROM files")
|
||||||
"ORDER BY path")
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read records: %w", err)
|
return nil, fmt.Errorf("read records: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
defer func() { _ = rows.Close() }()
|
defer func() { _ = rows.Close() }()
|
||||||
|
|
||||||
|
var recs []scanRec
|
||||||
|
|
||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
var (
|
var (
|
||||||
path []byte
|
path []byte
|
||||||
@@ -309,92 +225,19 @@ func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
|
|||||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
|
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
|
||||||
&r.content)
|
&r.content)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read record: %w", err)
|
return nil, fmt.Errorf("read record: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
r.path = string(path)
|
r.path = string(path)
|
||||||
fn(r)
|
recs = append(recs, r)
|
||||||
}
|
}
|
||||||
|
|
||||||
err = rows.Err()
|
err = rows.Err()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read records: %w", err)
|
return nil, fmt.Errorf("read records: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
return nil
|
return recs, nil
|
||||||
}
|
|
||||||
|
|
||||||
// dupeRowsSQL selects every record in a duplicate group, with the
|
|
||||||
// group's first path. A group is the records with a content hash that
|
|
||||||
// share a size, head, tail, and content, when there are two or more of
|
|
||||||
// them. The rows come in report order: groups by size descending, then
|
|
||||||
// by first path, and each group's paths ascending.
|
|
||||||
const dupeRowsSQL = `
|
|
||||||
SELECT g.first, f.path, f.size
|
|
||||||
FROM files AS f
|
|
||||||
JOIN (
|
|
||||||
SELECT size, head, tail, content, MIN(path) AS first
|
|
||||||
FROM files
|
|
||||||
WHERE content <> ''
|
|
||||||
GROUP BY size, head, tail, content
|
|
||||||
HAVING COUNT(*) > 1
|
|
||||||
) AS g USING (size, head, tail, content)
|
|
||||||
ORDER BY f.size DESC, g.first, f.path
|
|
||||||
`
|
|
||||||
|
|
||||||
// loadDupeRows streams the rows of dupeRowsSQL to fn and returns the
|
|
||||||
// number of records in the database. The count and the rows are read
|
|
||||||
// in one transaction, so they agree while a scan is committing. An
|
|
||||||
// error from fn stops the reading and is returned as it is.
|
|
||||||
func loadDupeRows(ctx context.Context, db *sql.DB,
|
|
||||||
fn func(first, path string, size int64) error,
|
|
||||||
) (int, error) {
|
|
||||||
// Everything goes through tx: the report connection is the only
|
|
||||||
// one, so a query on db would wait for tx forever.
|
|
||||||
tx, err := db.BeginTx(ctx, &sql.TxOptions{ReadOnly: true})
|
|
||||||
if err != nil {
|
|
||||||
return 0, fmt.Errorf("read records: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = tx.Rollback() }()
|
|
||||||
|
|
||||||
var records int
|
|
||||||
|
|
||||||
err = tx.QueryRowContext(ctx, "SELECT COUNT(*) FROM files").Scan(&records)
|
|
||||||
if err != nil {
|
|
||||||
return 0, fmt.Errorf("read records: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
rows, err := tx.QueryContext(ctx, dupeRowsSQL)
|
|
||||||
if err != nil {
|
|
||||||
return 0, fmt.Errorf("read records: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = rows.Close() }()
|
|
||||||
|
|
||||||
for rows.Next() {
|
|
||||||
var (
|
|
||||||
first, path []byte
|
|
||||||
size int64
|
|
||||||
)
|
|
||||||
|
|
||||||
err = rows.Scan(&first, &path, &size)
|
|
||||||
if err != nil {
|
|
||||||
return 0, fmt.Errorf("read record: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = fn(string(first), string(path), size)
|
|
||||||
if err != nil {
|
|
||||||
return 0, err
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
err = rows.Err()
|
|
||||||
if err != nil {
|
|
||||||
return 0, fmt.Errorf("read records: %w", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return records, nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// loadFileMeta streams every record's path, size, mtime, and whether
|
// loadFileMeta streams every record's path, size, mtime, and whether
|
||||||
|
|||||||
+24
-53
@@ -5,9 +5,9 @@ import (
|
|||||||
"database/sql"
|
"database/sql"
|
||||||
"errors"
|
"errors"
|
||||||
"fmt"
|
"fmt"
|
||||||
"os"
|
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"slices"
|
"slices"
|
||||||
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -72,8 +72,9 @@ func TestOpenScanDatabaseCreates(t *testing.T) {
|
|||||||
|
|
||||||
defer func() { _ = db.Close() }()
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
if recs := dbRecords(t, db); len(recs) != 0 {
|
recs, err := loadFileRows(t.Context(), db)
|
||||||
t.Fatalf("records = %v, want none", recs)
|
if err != nil || len(recs) != 0 {
|
||||||
|
t.Fatalf("loadFileRows = %v, %v; want empty, nil", recs, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -129,49 +130,6 @@ func TestOpenReportDatabaseOK(t *testing.T) {
|
|||||||
_ = db.Close()
|
_ = db.Close()
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestCloseScanDatabaseWhileReportOpen(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
// A report holding the database open stops scan from taking it out
|
|
||||||
// of WAL mode. The -wal and -shm files must then stay beside it, so
|
|
||||||
// that a later report still needs only read access.
|
|
||||||
path := testDBPath(t)
|
|
||||||
|
|
||||||
scanDB, err := openScanDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
reportDB, err := openReportDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
closeScanDatabase(t.Context(), scanDB, path)
|
|
||||||
|
|
||||||
_ = reportDB.Close()
|
|
||||||
|
|
||||||
_, err = os.Stat(path + "-wal")
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("no -wal left: the switch out of WAL mode was not "+
|
|
||||||
"stopped: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
makeReadOnly(t, path)
|
|
||||||
|
|
||||||
reportDB, err = openReportDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("openReportDatabase: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = reportDB.Close() }()
|
|
||||||
|
|
||||||
err = loadFileRows(t.Context(), reportDB, func(scanRec) {})
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("loadFileRows: %v", err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestApplyChangesRoundTrip(t *testing.T) {
|
func TestApplyChangesRoundTrip(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
@@ -194,8 +152,15 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
|||||||
t.Fatalf("applyChanges: %v", err)
|
t.Fatalf("applyChanges: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
// The records come back in path order, which is the order of recs.
|
got, err := loadFileRows(t.Context(), db)
|
||||||
got := dbRecords(t, db)
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
slices.SortFunc(got, func(a, b scanRec) int {
|
||||||
|
return strings.Compare(a.path, b.path)
|
||||||
|
})
|
||||||
|
|
||||||
if !slices.Equal(got, recs) {
|
if !slices.Equal(got, recs) {
|
||||||
t.Fatalf("rows = %+v, want %+v", got, recs)
|
t.Fatalf("rows = %+v, want %+v", got, recs)
|
||||||
}
|
}
|
||||||
@@ -212,7 +177,11 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
|||||||
t.Fatalf("applyChanges: %v", err)
|
t.Fatalf("applyChanges: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
got = dbRecords(t, db)
|
got, err = loadFileRows(t.Context(), db)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
if len(got) != 1 || got[0] != upd {
|
if len(got) != 1 || got[0] != upd {
|
||||||
t.Fatalf("rows = %+v, want just %+v", got, upd)
|
t.Fatalf("rows = %+v, want just %+v", got, upd)
|
||||||
}
|
}
|
||||||
@@ -241,8 +210,9 @@ func TestApplyChangesBatching(t *testing.T) {
|
|||||||
t.Fatalf("applyChanges: %v", err)
|
t.Fatalf("applyChanges: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
if got := dbRecords(t, db); len(got) != n {
|
got, err := loadFileRows(t.Context(), db)
|
||||||
t.Fatalf("records = %d, want %d", len(got), n)
|
if err != nil || len(got) != n {
|
||||||
|
t.Fatalf("loadFileRows = %d rows, %v; want %d", len(got), err, n)
|
||||||
}
|
}
|
||||||
|
|
||||||
deletes := make([]string, 0, n)
|
deletes := make([]string, 0, n)
|
||||||
@@ -256,7 +226,8 @@ func TestApplyChangesBatching(t *testing.T) {
|
|||||||
t.Fatalf("applyChanges deletes: %v", err)
|
t.Fatalf("applyChanges deletes: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
if got := dbRecords(t, db); len(got) != 0 {
|
got, err = loadFileRows(t.Context(), db)
|
||||||
t.Fatalf("records = %d, want 0", len(got))
|
if err != nil || len(got) != 0 {
|
||||||
|
t.Fatalf("loadFileRows = %d rows, %v; want 0", len(got), err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -5,8 +5,6 @@ go 1.25.7
|
|||||||
require (
|
require (
|
||||||
github.com/schollz/progressbar/v3 v3.19.1
|
github.com/schollz/progressbar/v3 v3.19.1
|
||||||
github.com/spf13/cobra v1.10.2
|
github.com/spf13/cobra v1.10.2
|
||||||
golang.org/x/sys v0.46.0
|
|
||||||
golang.org/x/term v0.44.0
|
|
||||||
modernc.org/sqlite v1.54.0
|
modernc.org/sqlite v1.54.0
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -20,6 +18,8 @@ require (
|
|||||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
||||||
github.com/rivo/uniseg v0.4.7 // indirect
|
github.com/rivo/uniseg v0.4.7 // indirect
|
||||||
github.com/spf13/pflag v1.0.9 // indirect
|
github.com/spf13/pflag v1.0.9 // indirect
|
||||||
|
golang.org/x/sys v0.46.0 // indirect
|
||||||
|
golang.org/x/term v0.44.0 // indirect
|
||||||
modernc.org/libc v1.74.1 // indirect
|
modernc.org/libc v1.74.1 // indirect
|
||||||
modernc.org/mathutil v1.7.1 // indirect
|
modernc.org/mathutil v1.7.1 // indirect
|
||||||
modernc.org/memory v1.11.0 // indirect
|
modernc.org/memory v1.11.0 // indirect
|
||||||
|
|||||||
@@ -57,27 +57,22 @@ var errNoSubcommand = errors.New("no subcommand")
|
|||||||
var Version = "dev"
|
var Version = "dev"
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
// Once the reader of a stdout pipe has gone, as in "sfdupes report |
|
os.Exit(run(os.Args[1:], os.Stderr))
|
||||||
// head", the Go runtime ends the process with SIGPIPE on the next
|
|
||||||
// write instead of returning an error (README "Error handling").
|
|
||||||
// Registering for SIGPIPE with os/signal would change that.
|
|
||||||
os.Exit(run(os.Args[1:], os.Stdout, os.Stderr))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// run executes args against the command tree and returns the process
|
// run executes args against the command tree and returns the process
|
||||||
// exit code. It is the program's single exit point: the subcommands
|
// exit code. It is the program's single exit point: the subcommands
|
||||||
// return their errors instead of exiting, so every deferred cleanup —
|
// return their errors instead of exiting, so every deferred cleanup —
|
||||||
// above all closing the database, which checkpoints the SQLite WAL —
|
// above all closing the database, which checkpoints the SQLite WAL —
|
||||||
// runs before the process ends. The report and trees subcommands write
|
// runs before the process ends.
|
||||||
// their data to stdout.
|
func run(args []string, stderr io.Writer) int {
|
||||||
func run(args []string, stdout, stderr io.Writer) int {
|
|
||||||
// A nil slice makes cobra fall back to os.Args, which would let a
|
// A nil slice makes cobra fall back to os.Args, which would let a
|
||||||
// test binary's own flags reach the command tree.
|
// test binary's own flags reach the command tree.
|
||||||
if args == nil {
|
if args == nil {
|
||||||
args = []string{}
|
args = []string{}
|
||||||
}
|
}
|
||||||
|
|
||||||
root := newRootCommand(stdout, stderr)
|
root := newRootCommand(stderr)
|
||||||
root.SetArgs(args)
|
root.SetArgs(args)
|
||||||
|
|
||||||
err := root.Execute()
|
err := root.Execute()
|
||||||
@@ -103,7 +98,7 @@ func run(args []string, stdout, stderr io.Writer) int {
|
|||||||
// newRootCommand builds the command tree. Everything on stdout is
|
// newRootCommand builds the command tree. Everything on stdout is
|
||||||
// machine-readable data; all human-facing output (help, usage, errors)
|
// machine-readable data; all human-facing output (help, usage, errors)
|
||||||
// goes to stderr.
|
// goes to stderr.
|
||||||
func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
func newRootCommand(stderr io.Writer) *cobra.Command {
|
||||||
root := &cobra.Command{
|
root := &cobra.Command{
|
||||||
Use: "sfdupes",
|
Use: "sfdupes",
|
||||||
Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
|
Short: "Find candidate duplicate files by size and head/tail/content SHA-256",
|
||||||
@@ -145,7 +140,7 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
|||||||
Short: "Read the scan database and print the file-level duplicates report",
|
Short: "Read the scan database and print the file-level duplicates report",
|
||||||
Args: cobra.NoArgs,
|
Args: cobra.NoArgs,
|
||||||
RunE: runE(func(ctx context.Context, _ []string) error {
|
RunE: runE(func(ctx context.Context, _ []string) error {
|
||||||
return runReport(ctx, stdout)
|
return runReport(ctx)
|
||||||
}),
|
}),
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -154,7 +149,7 @@ func newRootCommand(stdout, stderr io.Writer) *cobra.Command {
|
|||||||
Short: "Read the scan database and print the duplicate-tree report",
|
Short: "Read the scan database and print the duplicate-tree report",
|
||||||
Args: cobra.NoArgs,
|
Args: cobra.NoArgs,
|
||||||
RunE: runE(func(ctx context.Context, _ []string) error {
|
RunE: runE(func(ctx context.Context, _ []string) error {
|
||||||
return runTrees(ctx, stdout)
|
return runTrees(ctx)
|
||||||
}),
|
}),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+55
-376
@@ -44,61 +44,23 @@ func assertNoSidecars(t *testing.T, path string) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// makeReadOnly takes write permission away from the database at path,
|
// captureStdout redirects os.Stdout to a file for the rest of the test
|
||||||
// from any WAL sidecar beside it, and from their directory, as for a
|
// and returns a function reading back everything written to it. Only
|
||||||
// user reading a database that a root cron scan keeps. Root ignores
|
// machine-readable data belongs on stdout (README design goal 4), so
|
||||||
// file permissions, so it skips the test when run as root.
|
// the tests assert on it directly.
|
||||||
func makeReadOnly(t *testing.T, path string) {
|
func captureStdout(t *testing.T) func() string {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
if os.Geteuid() == 0 {
|
f, err := os.Create(filepath.Join(t.TempDir(), "stdout"))
|
||||||
t.Skip("root ignores file permissions")
|
|
||||||
}
|
|
||||||
|
|
||||||
err := os.Chmod(path, 0o400)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
|
|
||||||
for _, suffix := range walSuffixes {
|
saved := os.Stdout
|
||||||
err = os.Chmod(path+suffix, 0o400)
|
os.Stdout = f
|
||||||
if err != nil && !errors.Is(err, fs.ErrNotExist) {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
dir := filepath.Dir(path)
|
|
||||||
|
|
||||||
//nolint:gosec // reaching the database needs the search bit
|
|
||||||
err = os.Chmod(dir, 0o500)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Runs before t.TempDir's own cleanup, which must delete the files.
|
|
||||||
t.Cleanup(func() {
|
|
||||||
//nolint:gosec // removing the directory needs its search bit back
|
|
||||||
_ = os.Chmod(dir, 0o700)
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
// captureStderr redirects os.Stderr to a file for the rest of the test
|
|
||||||
// and returns a function reading back everything written to it. scan
|
|
||||||
// writes its warnings and summary straight to os.Stderr, not to the
|
|
||||||
// stderr writer run is given.
|
|
||||||
func captureStderr(t *testing.T) func() string {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
saved := os.Stderr
|
|
||||||
os.Stderr = f
|
|
||||||
|
|
||||||
t.Cleanup(func() {
|
t.Cleanup(func() {
|
||||||
os.Stderr = saved
|
os.Stdout = saved
|
||||||
|
|
||||||
_ = f.Close()
|
_ = f.Close()
|
||||||
})
|
})
|
||||||
@@ -129,13 +91,13 @@ func captureStderr(t *testing.T) func() string {
|
|||||||
// brokenDatabase writes a database that opens cleanly and passes the
|
// brokenDatabase writes a database that opens cleanly and passes the
|
||||||
// schema-version check but has no files table, so the first query
|
// schema-version check but has no files table, so the first query
|
||||||
// fails with the database already open: a fatal error on a path that
|
// fails with the database already open: a fatal error on a path that
|
||||||
// owns an open database. It closes the database the way scan does.
|
// owns an open database.
|
||||||
func brokenDatabase(t *testing.T) string {
|
func brokenDatabase(t *testing.T) string {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
path := testDBPath(t)
|
path := testDBPath(t)
|
||||||
|
|
||||||
db, err := openDB(path, scanParams)
|
db, err := openDB(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
@@ -146,7 +108,10 @@ func brokenDatabase(t *testing.T) string {
|
|||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
|
|
||||||
closeScanDatabase(t.Context(), db, path)
|
err = db.Close()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
return path
|
return path
|
||||||
}
|
}
|
||||||
@@ -179,10 +144,7 @@ func TestOpenDatabaseKeepsWALWhileOpen(t *testing.T) {
|
|||||||
|
|
||||||
func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
|
func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
|
||||||
// Every subcommand that owns an open database must close it when
|
// Every subcommand that owns an open database must close it when
|
||||||
// it fails: no os.Exit between the open and the return. The
|
// it fails: no os.Exit between the open and the return.
|
||||||
// sidecar check is evidence of the close only for scan: report and
|
|
||||||
// trees only read a database that is out of WAL mode, which leaves
|
|
||||||
// nothing on disk whether they close it or not.
|
|
||||||
cases := map[string][]string{
|
cases := map[string][]string{
|
||||||
cmdScan: {cmdScan},
|
cmdScan: {cmdScan},
|
||||||
cmdReport: {cmdReport},
|
cmdReport: {cmdReport},
|
||||||
@@ -198,15 +160,17 @@ func TestRunFatalAfterOpenClosesDatabase(t *testing.T) {
|
|||||||
args = append(args, t.TempDir())
|
args = append(args, t.TempDir())
|
||||||
}
|
}
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
code := run(args, &stdout, &stderr)
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
|
code := run(args, &stderr)
|
||||||
if code != exitFatal {
|
if code != exitFatal {
|
||||||
t.Errorf("run(%v) = %d, want %d", args, code, exitFatal)
|
t.Errorf("run(%v) = %d, want %d", args, code, exitFatal)
|
||||||
}
|
}
|
||||||
|
|
||||||
assertNoSidecars(t, path)
|
assertNoSidecars(t, path)
|
||||||
assertFatalOutput(t, stderr.String(), stdout.String())
|
assertFatalOutput(t, stderr.String(), stdout())
|
||||||
|
|
||||||
// Proof that the failure happened after the open: only a
|
// Proof that the failure happened after the open: only a
|
||||||
// query against the opened database can report this.
|
// query against the opened database can report this.
|
||||||
@@ -224,16 +188,18 @@ func TestRunMissingOperandIsFatalNotUsage(t *testing.T) {
|
|||||||
// must not dump the usage text.
|
// must not dump the usage text.
|
||||||
t.Setenv(databaseEnv, testDBPath(t))
|
t.Setenv(databaseEnv, testDBPath(t))
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
missing := filepath.Join(t.TempDir(), "nope")
|
missing := filepath.Join(t.TempDir(), "nope")
|
||||||
|
|
||||||
code := run([]string{cmdScan, missing}, &stdout, &stderr)
|
code := run([]string{cmdScan, missing}, &stderr)
|
||||||
if code != exitFatal {
|
if code != exitFatal {
|
||||||
t.Errorf("run(scan %s) = %d, want %d", missing, code, exitFatal)
|
t.Errorf("run(scan %s) = %d, want %d", missing, code, exitFatal)
|
||||||
}
|
}
|
||||||
|
|
||||||
assertFatalOutput(t, stderr.String(), stdout.String())
|
assertFatalOutput(t, stderr.String(), stdout())
|
||||||
}
|
}
|
||||||
|
|
||||||
// assertFatalOutput checks that a fatal error was reported the way
|
// assertFatalOutput checks that a fatal error was reported the way
|
||||||
@@ -277,9 +243,11 @@ func TestRunUsageErrors(t *testing.T) {
|
|||||||
// path that does not exist.
|
// path that does not exist.
|
||||||
t.Setenv(databaseEnv, testDBPath(t))
|
t.Setenv(databaseEnv, testDBPath(t))
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
code := run(tc.args, &stdout, &stderr)
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
|
code := run(tc.args, &stderr)
|
||||||
if code != exitUsage {
|
if code != exitUsage {
|
||||||
t.Errorf("run(%v) = %d, want %d", tc.args, code, exitUsage)
|
t.Errorf("run(%v) = %d, want %d", tc.args, code, exitUsage)
|
||||||
}
|
}
|
||||||
@@ -288,7 +256,7 @@ func TestRunUsageErrors(t *testing.T) {
|
|||||||
t.Errorf("stderr = %q, want %q", stderr.String(), tc.want)
|
t.Errorf("stderr = %q, want %q", stderr.String(), tc.want)
|
||||||
}
|
}
|
||||||
|
|
||||||
if got := stdout.String(); got != "" {
|
if got := stdout(); got != "" {
|
||||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
t.Errorf("stdout = %q, want nothing (data only)", got)
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
@@ -297,9 +265,9 @@ func TestRunUsageErrors(t *testing.T) {
|
|||||||
|
|
||||||
// TestRunHelpAndVersionSucceed checks that the two informational flags
|
// TestRunHelpAndVersionSucceed checks that the two informational flags
|
||||||
// exit 0 and keep their human-facing output on stderr.
|
// exit 0 and keep their human-facing output on stderr.
|
||||||
|
//
|
||||||
|
//nolint:paralleltest // captureStdout replaces the process-wide os.Stdout
|
||||||
func TestRunHelpAndVersionSucceed(t *testing.T) {
|
func TestRunHelpAndVersionSucceed(t *testing.T) {
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
assertHumanOutput(t, "--help")
|
assertHumanOutput(t, "--help")
|
||||||
assertHumanOutput(t, "--version")
|
assertHumanOutput(t, "--version")
|
||||||
}
|
}
|
||||||
@@ -310,9 +278,11 @@ func TestRunHelpAndVersionSucceed(t *testing.T) {
|
|||||||
func assertHumanOutput(t *testing.T, arg string) {
|
func assertHumanOutput(t *testing.T, arg string) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
code := run([]string{arg}, &stdout, &stderr)
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
|
code := run([]string{arg}, &stderr)
|
||||||
if code != exitOK {
|
if code != exitOK {
|
||||||
t.Errorf("run(%s) = %d, want %d", arg, code, exitOK)
|
t.Errorf("run(%s) = %d, want %d", arg, code, exitOK)
|
||||||
}
|
}
|
||||||
@@ -321,7 +291,7 @@ func assertHumanOutput(t *testing.T, arg string) {
|
|||||||
t.Errorf("run(%s) wrote nothing to stderr", arg)
|
t.Errorf("run(%s) wrote nothing to stderr", arg)
|
||||||
}
|
}
|
||||||
|
|
||||||
if got := stdout.String(); got != "" {
|
if got := stdout(); got != "" {
|
||||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
t.Errorf("stdout = %q, want nothing (data only)", got)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -350,31 +320,21 @@ func scanFixture(t *testing.T) []string {
|
|||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
|
|
||||||
scanOK(t, dir)
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
return dupes
|
stdout := captureStdout(t)
|
||||||
}
|
|
||||||
|
|
||||||
// scanOK runs scan over operands, fails the test unless it exits 0 with
|
code := run([]string{cmdScan, dir}, &stderr)
|
||||||
// nothing on stdout, and returns everything it printed to stderr.
|
|
||||||
func scanOK(t *testing.T, operands ...string) string {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
var stdout bytes.Buffer
|
|
||||||
|
|
||||||
stderr := captureStderr(t)
|
|
||||||
|
|
||||||
code := run(append([]string{cmdScan}, operands...), &stdout, os.Stderr)
|
|
||||||
if code != exitOK {
|
if code != exitOK {
|
||||||
t.Fatalf("run(scan %q) = %d, want %d; stderr: %s",
|
t.Fatalf("run(scan) = %d, want %d; stderr: %s",
|
||||||
operands, code, exitOK, stderr())
|
code, exitOK, stderr.String())
|
||||||
}
|
}
|
||||||
|
|
||||||
if got := stdout.String(); got != "" {
|
if got := stdout(); got != "" {
|
||||||
t.Errorf("scan stdout = %q, want nothing (data only)", got)
|
t.Errorf("scan stdout = %q, want nothing (data only)", got)
|
||||||
}
|
}
|
||||||
|
|
||||||
return stderr()
|
return dupes
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
||||||
@@ -385,119 +345,24 @@ func TestRunScanSucceedsDespiteWarnings(t *testing.T) {
|
|||||||
assertNoSidecars(t, path)
|
assertNoSidecars(t, path)
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRunScanSkipsSymlinkOperand(t *testing.T) {
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
dir := t.TempDir()
|
|
||||||
writeFile(t, dir, "target/sub/f", pattern(1, 10))
|
|
||||||
|
|
||||||
link := filepath.Join(dir, "link")
|
|
||||||
|
|
||||||
err := os.Symlink(filepath.Join(dir, "target"), link)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Scanning a directory through the symlink stores a record beneath
|
|
||||||
// the symlink's own path for a file beneath its target.
|
|
||||||
scanOK(t, filepath.Join(link, "sub"))
|
|
||||||
|
|
||||||
assertOperandSkipped(t, path, link, "symlink",
|
|
||||||
filepath.Join(link, "sub", "f"))
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunScanWalksOperandUnderSymlinkOperand(t *testing.T) {
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
dir := t.TempDir()
|
|
||||||
writeFile(t, dir, "target/sub/f", pattern(1, 10))
|
|
||||||
|
|
||||||
link := filepath.Join(dir, "link")
|
|
||||||
|
|
||||||
err := os.Symlink(filepath.Join(dir, "target"), link)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// link is dropped as a symlink, but link/sub must still be scanned,
|
|
||||||
// not dropped as lying under link.
|
|
||||||
scanOK(t, link, filepath.Join(link, "sub"))
|
|
||||||
|
|
||||||
db, err := openDB(path, reportParams)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
t.Cleanup(func() { _ = db.Close() })
|
|
||||||
|
|
||||||
recordByPath(t, dbRecords(t, db), filepath.Join(link, "sub", "f"))
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunScanSkipsZFSOperand(t *testing.T) {
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
zfs := filepath.Join(t.TempDir(), ".zfs")
|
|
||||||
snapshot := filepath.Join(zfs, "snapshot", "hourly")
|
|
||||||
f := writeFile(t, snapshot, "f", pattern(1, 10))
|
|
||||||
|
|
||||||
// An operand beneath a .zfs directory is walked, because it is not
|
|
||||||
// itself named .zfs.
|
|
||||||
scanOK(t, snapshot)
|
|
||||||
|
|
||||||
assertOperandSkipped(t, path, zfs, ".zfs directory", f)
|
|
||||||
}
|
|
||||||
|
|
||||||
// assertOperandSkipped scans operand alone and checks that it is skipped
|
|
||||||
// as kind: a warning naming it, one skip in the summary, exit 0, and the
|
|
||||||
// record for kept, which an earlier scan stored beneath operand, still
|
|
||||||
// in the database at dbPath.
|
|
||||||
func assertOperandSkipped(t *testing.T, dbPath, operand, kind,
|
|
||||||
kept string,
|
|
||||||
) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
stderr := scanOK(t, operand)
|
|
||||||
|
|
||||||
warning := "walk " + operand + ": skipping " + kind + " operand\n"
|
|
||||||
if !strings.Contains(stderr, warning) {
|
|
||||||
t.Errorf("stderr = %q, want %q", stderr, warning)
|
|
||||||
}
|
|
||||||
|
|
||||||
summary := "scan: 0 files seen (0 added, 0 updated, 0 removed, " +
|
|
||||||
"0 unchanged), 1 skipped\n"
|
|
||||||
if !strings.Contains(stderr, summary) {
|
|
||||||
t.Errorf("stderr = %q, want %q", stderr, summary)
|
|
||||||
}
|
|
||||||
|
|
||||||
db, err := openDB(dbPath, reportParams)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
t.Cleanup(func() { _ = db.Close() })
|
|
||||||
|
|
||||||
recordByPath(t, dbRecords(t, db), kept)
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunReportSucceeds(t *testing.T) {
|
func TestRunReportSucceeds(t *testing.T) {
|
||||||
path := testDBPath(t)
|
path := testDBPath(t)
|
||||||
t.Setenv(databaseEnv, path)
|
t.Setenv(databaseEnv, path)
|
||||||
|
|
||||||
dupes := scanFixture(t)
|
dupes := scanFixture(t)
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
code := run([]string{cmdReport}, &stdout, &stderr)
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
|
code := run([]string{cmdReport}, &stderr)
|
||||||
if code != exitOK {
|
if code != exitOK {
|
||||||
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
||||||
code, exitOK, stderr.String())
|
code, exitOK, stderr.String())
|
||||||
}
|
}
|
||||||
|
|
||||||
want := "first\tdupe\tsize\n" + dupes[0] + "\t" + dupes[1] + "\t300\n"
|
want := "first\tdupe\tsize\n" + dupes[0] + "\t" + dupes[1] + "\t300\n"
|
||||||
if got := stdout.String(); got != want {
|
if got := stdout(); got != want {
|
||||||
t.Errorf("stdout = %q, want %q", got, want)
|
t.Errorf("stdout = %q, want %q", got, want)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -510,9 +375,11 @@ func TestRunTreesSucceeds(t *testing.T) {
|
|||||||
|
|
||||||
dupes := scanFixture(t)
|
dupes := scanFixture(t)
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
var stderr bytes.Buffer
|
||||||
|
|
||||||
code := run([]string{cmdTrees}, &stdout, &stderr)
|
stdout := captureStdout(t)
|
||||||
|
|
||||||
|
code := run([]string{cmdTrees}, &stderr)
|
||||||
if code != exitOK {
|
if code != exitOK {
|
||||||
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
||||||
code, exitOK, stderr.String())
|
code, exitOK, stderr.String())
|
||||||
@@ -522,197 +389,9 @@ func TestRunTreesSucceeds(t *testing.T) {
|
|||||||
// trees of each other.
|
// trees of each other.
|
||||||
want := "first\tdupe\tfiles\tsize\n" +
|
want := "first\tdupe\tfiles\tsize\n" +
|
||||||
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) + "\t1\t300\n"
|
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) + "\t1\t300\n"
|
||||||
if got := stdout.String(); got != want {
|
if got := stdout(); got != want {
|
||||||
t.Errorf("stdout = %q, want %q", got, want)
|
t.Errorf("stdout = %q, want %q", got, want)
|
||||||
}
|
}
|
||||||
|
|
||||||
assertNoSidecars(t, path)
|
assertNoSidecars(t, path)
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRunReportsNeedOnlyReadAccess(t *testing.T) {
|
|
||||||
// README §Database: report and trees need only read access to the
|
|
||||||
// database file. With its directory read-only as well, SQLite
|
|
||||||
// cannot create any file beside it.
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
dupes := scanFixture(t)
|
|
||||||
assertNoSidecars(t, path)
|
|
||||||
makeReadOnly(t, path)
|
|
||||||
|
|
||||||
cases := map[string]string{
|
|
||||||
cmdReport: "first\tdupe\tsize\n" +
|
|
||||||
dupes[0] + "\t" + dupes[1] + "\t300\n",
|
|
||||||
cmdTrees: "first\tdupe\tfiles\tsize\n" +
|
|
||||||
filepath.Dir(dupes[0]) + "\t" + filepath.Dir(dupes[1]) +
|
|
||||||
"\t1\t300\n",
|
|
||||||
}
|
|
||||||
|
|
||||||
for name, want := range cases {
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{name}, &stdout, &stderr)
|
|
||||||
if code != exitOK {
|
|
||||||
t.Errorf("run(%s) = %d, want %d; stderr: %s",
|
|
||||||
name, code, exitOK, stderr.String())
|
|
||||||
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
if got := stdout.String(); got != want {
|
|
||||||
t.Errorf("%s stdout = %q, want %q", name, got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// holdScanLock takes the lock on the database at path, as a running
|
|
||||||
// scan does, and holds it until the test ends. It fails the test when
|
|
||||||
// the lock is already held.
|
|
||||||
func holdScanLock(t *testing.T, path string) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
lock, err := lockScanDatabase(path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatalf("lock %s: %v", path, err)
|
|
||||||
}
|
|
||||||
|
|
||||||
t.Cleanup(func() { _ = lock.Close() })
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunSecondScanFails(t *testing.T) {
|
|
||||||
// README §Database: while one scan holds the lock, a second scan
|
|
||||||
// fails at once, naming the lock file, without creating the
|
|
||||||
// database.
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
holdScanLock(t, path)
|
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{cmdScan, t.TempDir()}, &stdout, &stderr)
|
|
||||||
if code != exitFatal {
|
|
||||||
t.Errorf("run(scan) = %d, want %d", code, exitFatal)
|
|
||||||
}
|
|
||||||
|
|
||||||
want := "sfdupes: another scan is running (lock held on " +
|
|
||||||
path + ".lock)\n"
|
|
||||||
if got := stderr.String(); got != want {
|
|
||||||
t.Errorf("stderr = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
if got := stdout.String(); got != "" {
|
|
||||||
t.Errorf("stdout = %q, want nothing (data only)", got)
|
|
||||||
}
|
|
||||||
|
|
||||||
_, err := os.Stat(path)
|
|
||||||
if !errors.Is(err, fs.ErrNotExist) {
|
|
||||||
t.Errorf("stat %s = %v, want the database not created", path, err)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunScanReleasesLock(t *testing.T) {
|
|
||||||
// README §Database: a scan releases the lock however it ends.
|
|
||||||
t.Run("success", func(t *testing.T) {
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
scanFixture(t)
|
|
||||||
holdScanLock(t, path)
|
|
||||||
})
|
|
||||||
|
|
||||||
t.Run("fatal error", func(t *testing.T) {
|
|
||||||
path := brokenDatabase(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
code := run([]string{cmdScan, t.TempDir()}, io.Discard, io.Discard)
|
|
||||||
if code != exitFatal {
|
|
||||||
t.Fatalf("run(scan) = %d, want %d", code, exitFatal)
|
|
||||||
}
|
|
||||||
|
|
||||||
holdScanLock(t, path)
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunReportsDuringScan(t *testing.T) {
|
|
||||||
// README §Database: report and trees never take the lock, so they
|
|
||||||
// run while a scan holds it.
|
|
||||||
path := testDBPath(t)
|
|
||||||
t.Setenv(databaseEnv, path)
|
|
||||||
|
|
||||||
scanFixture(t)
|
|
||||||
holdScanLock(t, path)
|
|
||||||
|
|
||||||
for _, name := range []string{cmdReport, cmdTrees} {
|
|
||||||
var stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{name}, io.Discard, &stderr)
|
|
||||||
if code != exitOK {
|
|
||||||
t.Errorf("run(%s) = %d, want %d; stderr: %s",
|
|
||||||
name, code, exitOK, stderr.String())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunStdoutClosedIsFatal(t *testing.T) {
|
|
||||||
// README §Error handling: a stdout write failure exits 1, reported
|
|
||||||
// in one line on stderr.
|
|
||||||
for _, name := range []string{cmdReport, cmdTrees} {
|
|
||||||
t.Run(name, func(t *testing.T) {
|
|
||||||
t.Setenv(databaseEnv, testDBPath(t))
|
|
||||||
|
|
||||||
scanFixture(t)
|
|
||||||
|
|
||||||
stdout, err := os.Create(filepath.Join(t.TempDir(), "stdout"))
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = stdout.Close()
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
var stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{name}, stdout, &stderr)
|
|
||||||
if code != exitFatal {
|
|
||||||
t.Errorf("run(%s) = %d, want %d", name, code, exitFatal)
|
|
||||||
}
|
|
||||||
|
|
||||||
got := stderr.String()
|
|
||||||
if !strings.HasPrefix(got, "sfdupes: write stdout: ") ||
|
|
||||||
!strings.Contains(got, os.ErrClosed.Error()) ||
|
|
||||||
strings.Count(got, "\n") != 1 {
|
|
||||||
t.Errorf("stderr = %q, want one line reporting the "+
|
|
||||||
"failed stdout write", got)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// errWriteFailed is the error failingWriter returns.
|
|
||||||
var errWriteFailed = errors.New("write failed")
|
|
||||||
|
|
||||||
// failingWriter is a stdout that fails every write.
|
|
||||||
type failingWriter struct{}
|
|
||||||
|
|
||||||
func (failingWriter) Write([]byte) (int, error) { return 0, errWriteFailed }
|
|
||||||
|
|
||||||
func TestStdoutWriteErrorPropagates(t *testing.T) {
|
|
||||||
t.Setenv(databaseEnv, testDBPath(t))
|
|
||||||
|
|
||||||
scanFixture(t)
|
|
||||||
|
|
||||||
cases := map[string]func(context.Context, io.Writer) error{
|
|
||||||
cmdReport: runReport,
|
|
||||||
cmdTrees: runTrees,
|
|
||||||
}
|
|
||||||
|
|
||||||
for name, fn := range cases {
|
|
||||||
err := fn(t.Context(), failingWriter{})
|
|
||||||
if !errors.Is(err, errWriteFailed) {
|
|
||||||
t.Errorf("%s: error = %v, want %v", name, err, errWriteFailed)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
+16
-39
@@ -6,7 +6,6 @@ import (
|
|||||||
"time"
|
"time"
|
||||||
|
|
||||||
"github.com/schollz/progressbar/v3"
|
"github.com/schollz/progressbar/v3"
|
||||||
"golang.org/x/term"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// plainInterval is the minimum time between progress lines when stderr
|
// plainInterval is the minimum time between progress lines when stderr
|
||||||
@@ -25,23 +24,24 @@ const percentScale = 100
|
|||||||
|
|
||||||
// stderrIsTTY reports whether stderr is attached to a terminal.
|
// stderrIsTTY reports whether stderr is attached to a terminal.
|
||||||
func stderrIsTTY() bool {
|
func stderrIsTTY() bool {
|
||||||
return term.IsTerminal(int(os.Stderr.Fd()))
|
fi, err := os.Stderr.Stat()
|
||||||
|
if err != nil {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
return fi.Mode()&os.ModeCharDevice != 0
|
||||||
}
|
}
|
||||||
|
|
||||||
// progress renders one scan pass's progress on stderr. On a TTY it
|
// progress renders one scan pass's progress on stderr. On a TTY it
|
||||||
// delegates to the progressbar library (spinner style when the total is
|
// delegates to the progressbar library (spinner style when the total is
|
||||||
// unknown, full bar with count/percent/rate/elapsed/ETA otherwise). When
|
// unknown, full bar with count/percent/rate/elapsed/ETA otherwise). When
|
||||||
// stderr is not a TTY it emits no ANSI redraws: it prints a plain
|
// stderr is not a TTY it emits no ANSI redraws: it prints a plain
|
||||||
// one-line update as the pass starts, then no more often than every
|
// one-line update no more often than every plainInterval.
|
||||||
// plainInterval.
|
|
||||||
//
|
//
|
||||||
// All methods must be called from the main goroutine only. On a TTY
|
// All methods must be called from the main goroutine only. A nil
|
||||||
// the library also redraws a spinner from its own goroutine, several
|
// *progress is a valid no-display receiver: every method is a no-op,
|
||||||
// times a second, so its count and elapsed time stay current while a
|
// so batched database flushes during the streaming pass can reuse the
|
||||||
// pass waits for its next item. A nil *progress is a valid
|
// update-pass helpers without rendering anything.
|
||||||
// no-display receiver: every method is a no-op, so batched database
|
|
||||||
// flushes during the streaming pass can reuse the update-pass helpers
|
|
||||||
// without rendering anything.
|
|
||||||
type progress struct {
|
type progress struct {
|
||||||
label string
|
label string
|
||||||
total int64 // -1 when unknown (walk pass)
|
total int64 // -1 when unknown (walk pass)
|
||||||
@@ -53,22 +53,10 @@ type progress struct {
|
|||||||
|
|
||||||
func newProgress(label string, total int64) *progress {
|
func newProgress(label string, total int64) *progress {
|
||||||
p := &progress{label: label, total: total, start: time.Now()}
|
p := &progress{label: label, total: total, start: time.Now()}
|
||||||
if stderrIsTTY() {
|
if !stderrIsTTY() {
|
||||||
p.bar = newBar(label, total)
|
|
||||||
|
|
||||||
return p
|
return p
|
||||||
}
|
}
|
||||||
|
|
||||||
// Print the zero state at once: the first item may take minutes,
|
|
||||||
// and a pass must never look hung.
|
|
||||||
p.last = p.start
|
|
||||||
fmt.Fprintln(os.Stderr, p.plainLine())
|
|
||||||
|
|
||||||
return p
|
|
||||||
}
|
|
||||||
|
|
||||||
// newBar builds the TTY display for newProgress.
|
|
||||||
func newBar(label string, total int64) *progressbar.ProgressBar {
|
|
||||||
opts := []progressbar.Option{
|
opts := []progressbar.Option{
|
||||||
progressbar.OptionSetWriter(os.Stderr),
|
progressbar.OptionSetWriter(os.Stderr),
|
||||||
progressbar.OptionSetDescription(label),
|
progressbar.OptionSetDescription(label),
|
||||||
@@ -93,7 +81,9 @@ func newBar(label string, total int64) *progressbar.ProgressBar {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
return progressbar.NewOptions64(total, opts...)
|
p.bar = progressbar.NewOptions64(total, opts...)
|
||||||
|
|
||||||
|
return p
|
||||||
}
|
}
|
||||||
|
|
||||||
// increment records one completed item and refreshes the display.
|
// increment records one completed item and refreshes the display.
|
||||||
@@ -116,29 +106,16 @@ func (p *progress) increment() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// warnf prints a one-line warning to stderr without corrupting the bar.
|
// warnf prints a one-line warning to stderr without corrupting the bar.
|
||||||
// The whole message is escaped like a report's path columns, so a path
|
|
||||||
// holding a newline cannot split the warning.
|
|
||||||
func (p *progress) warnf(format string, args ...any) {
|
func (p *progress) warnf(format string, args ...any) {
|
||||||
if p == nil {
|
if p == nil {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
msg := escapePath(fmt.Sprintf(format, args...))
|
|
||||||
|
|
||||||
if p.bar != nil && p.total < 0 {
|
|
||||||
// The library also redraws a spinner from its own goroutine, so
|
|
||||||
// a direct write could land inside a redraw. The bar prints the
|
|
||||||
// warning itself, just before its next redraw.
|
|
||||||
_, _ = progressbar.Bprintln(p.bar, msg)
|
|
||||||
|
|
||||||
return
|
|
||||||
}
|
|
||||||
|
|
||||||
if p.bar != nil {
|
if p.bar != nil {
|
||||||
_ = p.bar.Clear()
|
_ = p.bar.Clear()
|
||||||
}
|
}
|
||||||
|
|
||||||
fmt.Fprintln(os.Stderr, msg)
|
fmt.Fprintf(os.Stderr, format+"\n", args...)
|
||||||
}
|
}
|
||||||
|
|
||||||
// finish terminates the pass's display.
|
// finish terminates the pass's display.
|
||||||
|
|||||||
@@ -1,161 +0,0 @@
|
|||||||
package main
|
|
||||||
|
|
||||||
import (
|
|
||||||
"os"
|
|
||||||
"path/filepath"
|
|
||||||
"strings"
|
|
||||||
"testing"
|
|
||||||
"time"
|
|
||||||
)
|
|
||||||
|
|
||||||
// spinnerIdle comfortably outlasts the 100ms interval at which the
|
|
||||||
// progressbar library redraws a spinner from its own goroutine.
|
|
||||||
const spinnerIdle = 500 * time.Millisecond
|
|
||||||
|
|
||||||
//nolint:paralleltest // replaces the process-wide os.Stderr
|
|
||||||
func TestStderrIsTTYFalseForNonTerminals(t *testing.T) {
|
|
||||||
r, pipe, err := os.Pipe()
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
regular, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
devNull, err := os.OpenFile(os.DevNull, os.O_WRONLY, 0)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
saved := os.Stderr
|
|
||||||
|
|
||||||
t.Cleanup(func() {
|
|
||||||
os.Stderr = saved
|
|
||||||
|
|
||||||
for _, f := range []*os.File{r, pipe, regular, devNull} {
|
|
||||||
_ = f.Close()
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
cases := map[string]*os.File{
|
|
||||||
"a pipe": pipe,
|
|
||||||
"a regular file": regular,
|
|
||||||
os.DevNull: devNull,
|
|
||||||
}
|
|
||||||
|
|
||||||
for name, f := range cases {
|
|
||||||
os.Stderr = f
|
|
||||||
|
|
||||||
if stderrIsTTY() {
|
|
||||||
t.Errorf("stderrIsTTY() = true with stderr on %s", name)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestNewProgressPrintsBeforeFirstItem checks that each pass shows its
|
|
||||||
// zero state the moment it starts when stderr is not a terminal, and
|
|
||||||
// that the next line still waits for plainInterval.
|
|
||||||
//
|
|
||||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
|
||||||
func TestNewProgressPrintsBeforeFirstItem(t *testing.T) {
|
|
||||||
stderr := captureStderr(t)
|
|
||||||
|
|
||||||
newProgress("walk", -1).increment()
|
|
||||||
newProgress("hash", 10).increment()
|
|
||||||
|
|
||||||
want := "walk: 0 files, elapsed 0s\n" +
|
|
||||||
"hash: [0/10] 0% 0 files/s elapsed 0s eta ?\n"
|
|
||||||
if got := stderr(); got != want {
|
|
||||||
t.Errorf("stderr = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// newWalkSpinner returns the walk pass's terminal display, writing to
|
|
||||||
// os.Stderr whether or not it is a terminal, and stops the library's
|
|
||||||
// redraws when the test ends.
|
|
||||||
func newWalkSpinner(t *testing.T) *progress {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
p := &progress{
|
|
||||||
label: "walk", total: -1, start: time.Now(),
|
|
||||||
bar: newBar("walk", -1),
|
|
||||||
}
|
|
||||||
t.Cleanup(p.finish)
|
|
||||||
|
|
||||||
return p
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestProgressWarningsOnOwnLines drives the terminal display of the walk
|
|
||||||
// pass through a run of warnings with no items between them, as when the
|
|
||||||
// walk meets many unreadable paths, for several of the spinner's
|
|
||||||
// redraws: every warning must land on a line of its own, never inside a
|
|
||||||
// redraw.
|
|
||||||
//
|
|
||||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
|
||||||
func TestProgressWarningsOnOwnLines(t *testing.T) {
|
|
||||||
stderr := captureStderr(t)
|
|
||||||
p := newWalkSpinner(t)
|
|
||||||
|
|
||||||
// No pause between warnings: one written straight to stderr is
|
|
||||||
// garbled only if a redraw lands while it is being written.
|
|
||||||
issued := 0
|
|
||||||
for start := time.Now(); time.Since(start) < spinnerIdle; issued++ {
|
|
||||||
p.warnf("warning")
|
|
||||||
}
|
|
||||||
|
|
||||||
// The spinner prints the warnings at its next redraw.
|
|
||||||
time.Sleep(spinnerIdle)
|
|
||||||
|
|
||||||
// A terminal shows each line as the text after its last carriage
|
|
||||||
// return.
|
|
||||||
shown := 0
|
|
||||||
|
|
||||||
for line := range strings.SplitSeq(stderr(), "\n") {
|
|
||||||
if !strings.Contains(line, "warning") {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
shown++
|
|
||||||
|
|
||||||
if text := line[strings.LastIndex(line, "\r")+1:]; text != "warning" {
|
|
||||||
t.Errorf("terminal shows %q, want %q", text, "warning")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if shown != issued {
|
|
||||||
t.Errorf("%d warning lines, want %d", shown, issued)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestSpinnerShowsCountAfterBurst checks that once a burst of items
|
|
||||||
// faster than the redraw limit is over, the walk display shows every
|
|
||||||
// item completed while it waits for the next one.
|
|
||||||
//
|
|
||||||
//nolint:paralleltest // captureStderr replaces the process-wide os.Stderr
|
|
||||||
func TestSpinnerShowsCountAfterBurst(t *testing.T) {
|
|
||||||
stderr := captureStderr(t)
|
|
||||||
p := newWalkSpinner(t)
|
|
||||||
|
|
||||||
for range 50 {
|
|
||||||
p.increment()
|
|
||||||
}
|
|
||||||
|
|
||||||
time.Sleep(spinnerIdle)
|
|
||||||
|
|
||||||
// A terminal shows the last frame drawn. The library starts each
|
|
||||||
// frame with a carriage return and erases the previous one with
|
|
||||||
// spaces first.
|
|
||||||
var shown string
|
|
||||||
|
|
||||||
for frame := range strings.SplitSeq(stderr(), "\r") {
|
|
||||||
if strings.TrimSpace(frame) != "" {
|
|
||||||
shown = frame
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if !strings.Contains(shown, "(50/-,") {
|
|
||||||
t.Errorf("terminal shows %q, want a count of 50", shown)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -4,8 +4,8 @@ import (
|
|||||||
"bufio"
|
"bufio"
|
||||||
"context"
|
"context"
|
||||||
"fmt"
|
"fmt"
|
||||||
"io"
|
|
||||||
"os"
|
"os"
|
||||||
|
"slices"
|
||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -28,59 +28,73 @@ type scanRec struct {
|
|||||||
path string
|
path string
|
||||||
}
|
}
|
||||||
|
|
||||||
// runReport implements the report subcommand: it prints the file-level
|
// loadRecords opens the database and reads every file record for the
|
||||||
// duplicates report as TSV on stdout. SQLite groups and orders the
|
// report and trees subcommands. Any database problem — including a
|
||||||
// records, and each row is written as it is read, so no group is held
|
// missing database — is fatal. The error is returned rather than
|
||||||
// in memory. It never touches the scanned filesystem; its only I/O is
|
// exiting, so that the deferred close — which checkpoints the SQLite
|
||||||
// the database (with SQLite's temporary sort file), stdout, and stderr.
|
// WAL — always runs; the database is closed before the caller formats
|
||||||
// Any database problem, including a missing database, is fatal.
|
// its output, so it stays closed even if that output fails.
|
||||||
func runReport(ctx context.Context, stdout io.Writer) error {
|
func loadRecords(ctx context.Context) ([]scanRec, error) {
|
||||||
dbPath := databasePath()
|
dbPath := databasePath()
|
||||||
|
|
||||||
db, err := openReportDatabase(ctx, dbPath)
|
db, err := openReportDatabase(ctx, dbPath)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
defer func() { _ = db.Close() }()
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
out := bufio.NewWriterSize(stdout, ioBufSize)
|
recs, err := loadFileRows(ctx, db)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("database %s: %w", dbPath, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return recs, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// dupeGroup is one set of candidate-duplicate files: identical size,
|
||||||
|
// head hash, tail hash, and content hash. paths is sorted
|
||||||
|
// lexicographically; the first entry is the group's "first", the rest
|
||||||
|
// are dupes.
|
||||||
|
type dupeGroup struct {
|
||||||
|
size int64
|
||||||
|
paths []string
|
||||||
|
}
|
||||||
|
|
||||||
|
// runReport implements the report subcommand: it reads every record
|
||||||
|
// from the database and prints the file-level duplicates report as TSV
|
||||||
|
// on stdout. It never touches the scanned filesystem; its only I/O is
|
||||||
|
// the database, stdout, and stderr.
|
||||||
|
func runReport(ctx context.Context) error {
|
||||||
|
recs, err := loadRecords(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
dupes := collectDupeGroups(recs)
|
||||||
|
|
||||||
|
out := bufio.NewWriterSize(os.Stdout, ioBufSize)
|
||||||
|
|
||||||
_, err = fmt.Fprintln(out, "first\tdupe\tsize")
|
_, err = fmt.Fprintln(out, "first\tdupe\tsize")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("write stdout: %w", err)
|
return fmt.Errorf("write stdout: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
var (
|
dupeFiles := 0
|
||||||
groups, dupeFiles int
|
|
||||||
reclaimable int64
|
|
||||||
writeErr error
|
|
||||||
)
|
|
||||||
|
|
||||||
records, err := loadDupeRows(ctx, db,
|
var reclaimable int64
|
||||||
func(first, path string, size int64) error {
|
|
||||||
// A group's first path is its first row; every other
|
|
||||||
// path is a dupe.
|
|
||||||
if path == first {
|
|
||||||
groups++
|
|
||||||
|
|
||||||
return nil
|
for _, g := range dupes {
|
||||||
|
for _, p := range g.paths[1:] {
|
||||||
|
_, err = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
||||||
|
g.paths[0], p, g.size)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("write stdout: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
_, writeErr = fmt.Fprintf(out, "%s\t%s\t%d\n",
|
|
||||||
escapePath(first), escapePath(path), size)
|
|
||||||
dupeFiles++
|
dupeFiles++
|
||||||
reclaimable += size
|
reclaimable += g.size
|
||||||
|
}
|
||||||
return writeErr
|
|
||||||
})
|
|
||||||
|
|
||||||
if writeErr != nil {
|
|
||||||
return fmt.Errorf("write stdout: %w", writeErr)
|
|
||||||
}
|
|
||||||
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("database %s: %w", dbPath, err)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
err = out.Flush()
|
err = out.Flush()
|
||||||
@@ -91,24 +105,55 @@ func runReport(ctx context.Context, stdout io.Writer) error {
|
|||||||
fmt.Fprintf(os.Stderr,
|
fmt.Fprintf(os.Stderr,
|
||||||
"report: %d records read, %d duplicate groups, %d dupe files, "+
|
"report: %d records read, %d duplicate groups, %d dupe files, "+
|
||||||
"%s reclaimable\n",
|
"%s reclaimable\n",
|
||||||
records, groups, dupeFiles, humanBytes(reclaimable))
|
len(recs), len(dupes), dupeFiles, humanBytes(reclaimable))
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// escapePath returns a path as it is written in a report column (README
|
// collectDupeGroups groups records by signature and returns every group
|
||||||
// "Report output format"): a backslash, tab, newline or carriage return
|
// with two or more paths, each group's paths sorted lexicographically,
|
||||||
// becomes \\, \t, \n or \r, and every other byte is kept as it is.
|
// groups ordered by size descending then by first path ascending.
|
||||||
// Grouping and sorting use the raw path, never this form.
|
func collectDupeGroups(recs []scanRec) []dupeGroup {
|
||||||
func escapePath(p string) string {
|
groups := make(map[fileSig][]string)
|
||||||
// Most paths need no escaping; skip building a replacer for them.
|
|
||||||
if !strings.ContainsAny(p, "\\\t\n\r") {
|
for _, r := range recs {
|
||||||
return p
|
// A record without a content hash has unknown content and is
|
||||||
|
// never reported as a duplicate (README "Database").
|
||||||
|
if r.content == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
k := fileSig{
|
||||||
|
size: r.size, head: r.head, tail: r.tail, content: r.content,
|
||||||
|
}
|
||||||
|
groups[k] = append(groups[k], r.path)
|
||||||
}
|
}
|
||||||
|
|
||||||
return strings.NewReplacer(
|
var dupes []dupeGroup
|
||||||
`\`, `\\`, "\t", `\t`, "\n", `\n`, "\r", `\r`,
|
|
||||||
).Replace(p)
|
for k, paths := range groups {
|
||||||
|
if len(paths) < minGroupSize {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
slices.Sort(paths)
|
||||||
|
dupes = append(dupes, dupeGroup{size: k.size, paths: paths})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Biggest reclaimable space first; ties broken by first path.
|
||||||
|
slices.SortFunc(dupes, func(a, b dupeGroup) int {
|
||||||
|
if a.size != b.size {
|
||||||
|
if a.size > b.size {
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
return strings.Compare(a.paths[0], b.paths[0])
|
||||||
|
})
|
||||||
|
|
||||||
|
return dupes
|
||||||
}
|
}
|
||||||
|
|
||||||
// humanBytes formats a byte count in human units (binary prefixes).
|
// humanBytes formats a byte count in human units (binary prefixes).
|
||||||
|
|||||||
+15
-260
@@ -1,254 +1,11 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
|
||||||
"database/sql"
|
|
||||||
"errors"
|
|
||||||
"fmt"
|
|
||||||
"io"
|
|
||||||
"os"
|
|
||||||
"path/filepath"
|
|
||||||
"slices"
|
"slices"
|
||||||
"strings"
|
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
|
|
||||||
// awkwardDir is a directory name holding every byte the reports escape.
|
func TestCollectDupeGroups(t *testing.T) {
|
||||||
const awkwardDir = "/d/\tone\ntwo\rthree\\four"
|
|
||||||
|
|
||||||
// awkwardPairRecs is a duplicate pair in sibling directories /d/A and
|
|
||||||
// awkwardDir. A raw tab sorts before "A" but its escaped form `\t`
|
|
||||||
// sorts after it, so awkwardDir coming first shows that sorting uses
|
|
||||||
// the raw path.
|
|
||||||
func awkwardPairRecs() []scanRec {
|
|
||||||
return []scanRec{
|
|
||||||
{size: 5, head: "h", tail: "t", content: "c", path: "/d/A/f"},
|
|
||||||
{size: 5, head: "h", tail: "t", content: "c", path: awkwardDir + "/f"},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// seedDatabase writes recs into a fresh database and returns its path.
|
|
||||||
func seedDatabase(t *testing.T, recs []scanRec) string {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
path := testDBPath(t)
|
|
||||||
|
|
||||||
db, err := openScanDatabase(t.Context(), path)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = applyChanges(t.Context(), db, recs, nil, nil)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
err = db.Close()
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return path
|
|
||||||
}
|
|
||||||
|
|
||||||
// dupeGroup is one duplicate group as report reads it: the size, and
|
|
||||||
// the paths in report order, first path first.
|
|
||||||
type dupeGroup struct {
|
|
||||||
size int64
|
|
||||||
paths []string
|
|
||||||
}
|
|
||||||
|
|
||||||
// dupeGroups returns the duplicate groups report reads from db, in
|
|
||||||
// report order.
|
|
||||||
func dupeGroups(t *testing.T, db *sql.DB) []dupeGroup {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
var groups []dupeGroup
|
|
||||||
|
|
||||||
_, err := loadDupeRows(t.Context(), db,
|
|
||||||
func(first, path string, size int64) error {
|
|
||||||
if path == first {
|
|
||||||
groups = append(groups, dupeGroup{size: size})
|
|
||||||
}
|
|
||||||
|
|
||||||
g := &groups[len(groups)-1]
|
|
||||||
g.paths = append(g.paths, path)
|
|
||||||
|
|
||||||
return nil
|
|
||||||
})
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return groups
|
|
||||||
}
|
|
||||||
|
|
||||||
// dupeGroupsOf writes recs into a fresh database and returns the
|
|
||||||
// duplicate groups report reads from it.
|
|
||||||
func dupeGroupsOf(t *testing.T, recs []scanRec) []dupeGroup {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
db := openTestDB(t)
|
|
||||||
|
|
||||||
err := applyChanges(t.Context(), db, recs, nil, nil)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return dupeGroups(t, db)
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunReportEscapesPaths(t *testing.T) {
|
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{cmdReport}, &stdout, &stderr)
|
|
||||||
if code != exitOK {
|
|
||||||
t.Fatalf("run(report) = %d, want %d; stderr: %s",
|
|
||||||
code, exitOK, stderr.String())
|
|
||||||
}
|
|
||||||
|
|
||||||
want := "first\tdupe\tsize\n" +
|
|
||||||
`/d/\tone\ntwo\rthree\\four/f` + "\t/d/A/f\t5\n"
|
|
||||||
if got := stdout.String(); got != want {
|
|
||||||
t.Errorf("stdout = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestReportStdoutFailsWhileReading(t *testing.T) {
|
|
||||||
// Each row holds two paths longer than dir, so the report is more
|
|
||||||
// than twice the stdout buffer and stdout fails while rows are
|
|
||||||
// still being read, not at the final flush.
|
|
||||||
dir := "/" + strings.Repeat("d", 4096)
|
|
||||||
|
|
||||||
recs := make([]scanRec, ioBufSize/len(dir))
|
|
||||||
for i := range recs {
|
|
||||||
recs[i] = scanRec{
|
|
||||||
size: 1, head: "h", tail: "t", content: "c",
|
|
||||||
path: fmt.Sprintf("%s/%d", dir, i),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, recs))
|
|
||||||
|
|
||||||
err := runReport(t.Context(), failingWriter{})
|
|
||||||
if !errors.Is(err, errWriteFailed) ||
|
|
||||||
!strings.HasPrefix(err.Error(), "write stdout: ") {
|
|
||||||
t.Errorf("error = %v, want write stdout: %v", err, errWriteFailed)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunReportsIgnoreInsertionOrder(t *testing.T) {
|
|
||||||
// README §Constraints: identical database contents give identical
|
|
||||||
// output, whatever order the records were inserted in.
|
|
||||||
recs := append(smokeTreeRecs(), awkwardPairRecs()...)
|
|
||||||
recs = append(recs,
|
|
||||||
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/2"},
|
|
||||||
scanRec{size: 50, head: "b", tail: "b", content: "b", path: "/y/1"},
|
|
||||||
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/2"},
|
|
||||||
scanRec{size: 50, head: "a", tail: "a", content: "a", path: "/x/1"},
|
|
||||||
scanRec{size: 50, path: "/x/unhashed"},
|
|
||||||
)
|
|
||||||
|
|
||||||
reversed := slices.Clone(recs)
|
|
||||||
slices.Reverse(reversed)
|
|
||||||
|
|
||||||
for _, name := range []string{cmdReport, cmdTrees} {
|
|
||||||
t.Run(name, func(t *testing.T) {
|
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, recs))
|
|
||||||
|
|
||||||
forward := runStdout(t, name)
|
|
||||||
|
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, reversed))
|
|
||||||
|
|
||||||
backward := runStdout(t, name)
|
|
||||||
|
|
||||||
if strings.Count(forward, "\n") < 3 {
|
|
||||||
t.Errorf("stdout = %q, want at least two rows", forward)
|
|
||||||
}
|
|
||||||
|
|
||||||
if forward != backward {
|
|
||||||
t.Errorf("stdout depends on insertion order: %q vs %q",
|
|
||||||
forward, backward)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// runStdout runs the subcommand name and returns its stdout, failing
|
|
||||||
// the test unless it succeeds.
|
|
||||||
func runStdout(t *testing.T, name string) string {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{name}, &stdout, &stderr)
|
|
||||||
if code != exitOK {
|
|
||||||
t.Fatalf("run(%s) = %d, want %d; stderr: %s",
|
|
||||||
name, code, exitOK, stderr.String())
|
|
||||||
}
|
|
||||||
|
|
||||||
return stdout.String()
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestEscapePath(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
cases := map[string]string{
|
|
||||||
"/srv/plain": "/srv/plain",
|
|
||||||
"/a\tb": `/a\tb`,
|
|
||||||
"/a\nb": `/a\nb`,
|
|
||||||
"/a\rb": `/a\rb`,
|
|
||||||
`/a\b`: `/a\\b`,
|
|
||||||
`/a\tb`: `/a\\tb`,
|
|
||||||
"/not-utf8\xff": "/not-utf8\xff",
|
|
||||||
}
|
|
||||||
for in, want := range cases {
|
|
||||||
if got := escapePath(in); got != want {
|
|
||||||
t.Errorf("escapePath(%q) = %q, want %q", in, got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// TestWarnfEscapes checks that a warning naming a path that holds a
|
|
||||||
// newline is still one line.
|
|
||||||
//
|
|
||||||
//nolint:paralleltest // replaces the process-wide os.Stderr
|
|
||||||
func TestWarnfEscapes(t *testing.T) {
|
|
||||||
f, err := os.Create(filepath.Join(t.TempDir(), "stderr"))
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
saved := os.Stderr
|
|
||||||
os.Stderr = f
|
|
||||||
|
|
||||||
t.Cleanup(func() {
|
|
||||||
os.Stderr = saved
|
|
||||||
|
|
||||||
_ = f.Close()
|
|
||||||
})
|
|
||||||
|
|
||||||
(&progress{}).warnf("stat %s: %s", "/d/a\nb", "gone")
|
|
||||||
|
|
||||||
_, err = f.Seek(0, io.SeekStart)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
got, err := io.ReadAll(f)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
want := `stat /d/a\nb: gone` + "\n"
|
|
||||||
if string(got) != want {
|
|
||||||
t.Errorf("warning = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestDupeGroups(t *testing.T) {
|
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
@@ -263,7 +20,7 @@ func TestDupeGroups(t *testing.T) {
|
|||||||
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
{size: 7, head: "u", tail: "u", content: "u", path: "/lonely"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := dupeGroupsOf(t, recs)
|
groups := collectDupeGroups(recs)
|
||||||
if len(groups) != 2 {
|
if len(groups) != 2 {
|
||||||
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
||||||
}
|
}
|
||||||
@@ -281,7 +38,7 @@ func TestDupeGroups(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestDupeGroupsContentSeparates(t *testing.T) {
|
func TestCollectDupeGroupsContentSeparates(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// Same size, head, and tail, but different content hashes: the final
|
// Same size, head, and tail, but different content hashes: the final
|
||||||
@@ -296,7 +53,7 @@ func TestDupeGroupsContentSeparates(t *testing.T) {
|
|||||||
{size: 100, head: "h", tail: "t", path: "/e"},
|
{size: 100, head: "h", tail: "t", path: "/e"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := dupeGroupsOf(t, recs)
|
groups := collectDupeGroups(recs)
|
||||||
if len(groups) != 1 {
|
if len(groups) != 1 {
|
||||||
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
t.Fatalf("len(groups) = %d, want 1 (only the matching content)",
|
||||||
len(groups))
|
len(groups))
|
||||||
@@ -307,7 +64,7 @@ func TestDupeGroupsContentSeparates(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestDupeGroupsMtimeExcluded(t *testing.T) {
|
func TestCollectDupeGroupsMtimeExcluded(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// mtime is informational only; records differing only in mtime
|
// mtime is informational only; records differing only in mtime
|
||||||
@@ -317,25 +74,23 @@ func TestDupeGroupsMtimeExcluded(t *testing.T) {
|
|||||||
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := dupeGroupsOf(t, recs)
|
groups := collectDupeGroups(recs)
|
||||||
if len(groups) != 1 {
|
if len(groups) != 1 {
|
||||||
t.Fatalf("len(groups) = %d, want 1", len(groups))
|
t.Fatalf("len(groups) = %d, want 1", len(groups))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestDupeGroupsTieBreak(t *testing.T) {
|
func TestCollectDupeGroupsTieBreak(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
// The hashes sort opposite to the first paths, so ordering the
|
|
||||||
// groups by hash instead of by first path fails this test.
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/2"},
|
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/2"},
|
||||||
{size: 50, head: "a", tail: "a", content: "a", path: "/beta/1"},
|
{size: 50, head: "b", tail: "b", content: "b", path: "/beta/1"},
|
||||||
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/2"},
|
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/2"},
|
||||||
{size: 50, head: "b", tail: "b", content: "b", path: "/alpha/1"},
|
{size: 50, head: "a", tail: "a", content: "a", path: "/alpha/1"},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := dupeGroupsOf(t, recs)
|
groups := collectDupeGroups(recs)
|
||||||
if len(groups) != 2 {
|
if len(groups) != 2 {
|
||||||
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
t.Fatalf("len(groups) = %d, want 2", len(groups))
|
||||||
}
|
}
|
||||||
@@ -347,7 +102,7 @@ func TestDupeGroupsTieBreak(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestDupeGroupsDeterministic(t *testing.T) {
|
func TestCollectDupeGroupsDeterministic(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
@@ -357,12 +112,12 @@ func TestDupeGroupsDeterministic(t *testing.T) {
|
|||||||
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
{size: 2, head: "b", tail: "b", content: "b", path: "/q/2"},
|
||||||
}
|
}
|
||||||
|
|
||||||
forward := dupeGroupsOf(t, recs)
|
forward := collectDupeGroups(recs)
|
||||||
|
|
||||||
reversed := slices.Clone(recs)
|
reversed := slices.Clone(recs)
|
||||||
slices.Reverse(reversed)
|
slices.Reverse(reversed)
|
||||||
|
|
||||||
backward := dupeGroupsOf(t, reversed)
|
backward := collectDupeGroups(reversed)
|
||||||
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
|
if !slices.EqualFunc(forward, backward, func(a, b dupeGroup) bool {
|
||||||
return a.size == b.size && slices.Equal(a.paths, b.paths)
|
return a.size == b.size && slices.Equal(a.paths, b.paths)
|
||||||
}) {
|
}) {
|
||||||
|
|||||||
@@ -85,13 +85,10 @@ type fileMeta struct {
|
|||||||
// least one other file shares are ever hashed: a size-unique file
|
// least one other file shares are ever hashed: a size-unique file
|
||||||
// cannot be a duplicate. A file of headTailMin or more gets its content
|
// cannot be a duplicate. A file of headTailMin or more gets its content
|
||||||
// hash only when its size, head, and tail match another file's. Flag
|
// hash only when its size, head, and tail match another file's. Flag
|
||||||
// parsing and the at-least-one-operand check are done by cobra. The
|
// parsing and the at-least-one-operand check are done by cobra. Errors
|
||||||
// scan holds the lock on the database for its whole run, so a second
|
// are returned rather than exiting, so that the deferred close — which
|
||||||
// scan fails before it walks the filesystem or opens the database.
|
// checkpoints the SQLite WAL — always runs. Cancelling ctx unwinds the
|
||||||
// Errors are returned rather than exiting, so that the deferred close —
|
// worker pools and aborts the scan with the context's error.
|
||||||
// which takes the database out of WAL mode — always runs, and the lock
|
|
||||||
// is released after it. Cancelling ctx unwinds the worker pools and
|
|
||||||
// aborts the scan with the context's error.
|
|
||||||
func runScan(ctx context.Context, roots []string, workers int,
|
func runScan(ctx context.Context, roots []string, workers int,
|
||||||
oneFS bool,
|
oneFS bool,
|
||||||
) error {
|
) error {
|
||||||
@@ -106,19 +103,12 @@ func runScan(ctx context.Context, roots []string, workers int,
|
|||||||
|
|
||||||
dbPath := databasePath()
|
dbPath := databasePath()
|
||||||
|
|
||||||
lock, err := lockScanDatabase(dbPath)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
|
|
||||||
defer func() { _ = lock.Close() }()
|
|
||||||
|
|
||||||
db, err := openScanDatabase(ctx, dbPath)
|
db, err := openScanDatabase(ctx, dbPath)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
defer closeScanDatabase(ctx, db, dbPath)
|
defer func() { _ = db.Close() }()
|
||||||
|
|
||||||
st, err := syncScan(ctx, db, roots, workers, oneFS)
|
st, err := syncScan(ctx, db, roots, workers, oneFS)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -220,17 +210,13 @@ type scanState struct {
|
|||||||
// in the content hash of every record of headTailMin or more whose
|
// in the content hash of every record of headTailMin or more whose
|
||||||
// size, head, and tail match another record's). Records outside the
|
// size, head, and tail match another record's). Records outside the
|
||||||
// roots are never touched, except that the content phase fills in
|
// roots are never touched, except that the content phase fills in
|
||||||
// their content hash. Operands the walk cannot start from are dropped
|
// their content hash.
|
||||||
// first, so the records beneath them count as outside the roots unless
|
|
||||||
// they lie under another root.
|
|
||||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||||
workers int, oneFS bool,
|
workers int, oneFS bool,
|
||||||
) (scanStats, error) {
|
) (scanStats, error) {
|
||||||
s := &scanState{db: db}
|
roots = pruneRoots(roots)
|
||||||
|
|
||||||
// Types are checked before pruning so that an operand under a
|
s := &scanState{db: db}
|
||||||
// dropped one is still scanned, not dropped as lying under it.
|
|
||||||
roots = pruneRoots(s.walkableRoots(roots))
|
|
||||||
|
|
||||||
err := s.loadIndex(ctx, roots)
|
err := s.loadIndex(ctx, roots)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -267,35 +253,6 @@ func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
|||||||
return s.st, s.contentPhase(ctx, workers)
|
return s.st, s.contentPhase(ctx, workers)
|
||||||
}
|
}
|
||||||
|
|
||||||
// walkableRoots returns the operands the walk can start from: regular
|
|
||||||
// files, and directories not named .zfs. Every other operand is warned
|
|
||||||
// about, counted as skipped, and dropped. A dropped operand is no
|
|
||||||
// longer a root, so the records stored beneath it count as outside the
|
|
||||||
// roots and are not deleted as unverified, unless it lies under another
|
|
||||||
// root. An operand that fails lstat here is kept, and the walk warns
|
|
||||||
// about it.
|
|
||||||
func (s *scanState) walkableRoots(roots []string) []string {
|
|
||||||
kept := make([]string, 0, len(roots))
|
|
||||||
|
|
||||||
for _, root := range roots {
|
|
||||||
fi, err := os.Lstat(root)
|
|
||||||
if err == nil {
|
|
||||||
warn := operandWarning(root, fi)
|
|
||||||
if warn != "" {
|
|
||||||
s.st.skipped++
|
|
||||||
|
|
||||||
fmt.Fprintln(os.Stderr, escapePath(warn))
|
|
||||||
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
kept = append(kept, root)
|
|
||||||
}
|
|
||||||
|
|
||||||
return kept
|
|
||||||
}
|
|
||||||
|
|
||||||
// loadIndex indexes the database records under the scan roots for
|
// loadIndex indexes the database records under the scan roots for
|
||||||
// change detection and collects the sizes of every record outside
|
// change detection and collects the sizes of every record outside
|
||||||
// them: out-of-scope records join the size census so a scanned file
|
// them: out-of-scope records join the size census so a scanned file
|
||||||
@@ -832,43 +789,11 @@ func sendEvent(ctx context.Context, events chan<- walkEvent,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// operandWarning returns the one-line warning for an operand the walk
|
|
||||||
// does not start from, naming the path and what it is, or "" for one it
|
|
||||||
// does: a regular file, or a directory not named .zfs. Symlinks are
|
|
||||||
// never followed, including as operands.
|
|
||||||
func operandWarning(root string, fi fs.FileInfo) string {
|
|
||||||
var kind string
|
|
||||||
|
|
||||||
switch mode := fi.Mode(); {
|
|
||||||
case mode.IsRegular():
|
|
||||||
return ""
|
|
||||||
case mode.IsDir():
|
|
||||||
if filepath.Base(root) != ".zfs" {
|
|
||||||
return ""
|
|
||||||
}
|
|
||||||
|
|
||||||
kind = ".zfs directory"
|
|
||||||
case mode&fs.ModeSymlink != 0:
|
|
||||||
kind = "symlink"
|
|
||||||
case mode&fs.ModeSocket != 0:
|
|
||||||
kind = "socket"
|
|
||||||
case mode&fs.ModeNamedPipe != 0:
|
|
||||||
kind = "FIFO"
|
|
||||||
case mode&fs.ModeDevice != 0:
|
|
||||||
kind = "device node"
|
|
||||||
default:
|
|
||||||
kind = "non-regular file"
|
|
||||||
}
|
|
||||||
|
|
||||||
return fmt.Sprintf("walk %s: skipping %s operand", root, kind)
|
|
||||||
}
|
|
||||||
|
|
||||||
// seedRoot turns one PATH operand into the walk's starting state: a
|
// seedRoot turns one PATH operand into the walk's starting state: a
|
||||||
// regular-file operand is statted and emitted directly, and a directory
|
// regular-file operand is statted and emitted directly, a directory
|
||||||
// operand becomes an initial job. walkableRoots has already dropped
|
// operand becomes an initial job, and a symlink or other non-regular
|
||||||
// every other operand. One that has changed into something else since
|
// operand yields nothing (symlinks are never followed, including as
|
||||||
// is warned about and skipped here; it is still a root, so the records
|
// operands).
|
||||||
// stored beneath it are deleted as unverified.
|
|
||||||
func seedRoot(ctx context.Context, root string,
|
func seedRoot(ctx context.Context, root string,
|
||||||
events chan<- walkEvent,
|
events chan<- walkEvent,
|
||||||
) []dirJob {
|
) []dirJob {
|
||||||
@@ -882,30 +807,30 @@ func seedRoot(ctx context.Context, root string,
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
warn := operandWarning(root, fi)
|
switch {
|
||||||
if warn != "" {
|
case fi.IsDir():
|
||||||
sendEvent(ctx, events, walkEvent{warn: warn, fail: true})
|
if filepath.Base(root) == ".zfs" {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
if fi.IsDir() {
|
|
||||||
dev, ok := deviceOfInfo(fi)
|
dev, ok := deviceOfInfo(fi)
|
||||||
|
|
||||||
return []dirJob{{path: root, rootDev: dev, rootDevOK: ok}}
|
return []dirJob{{path: root, rootDev: dev, rootDevOK: ok}}
|
||||||
|
case fi.Mode().IsRegular():
|
||||||
|
dev, ino := inodeOfInfo(fi)
|
||||||
|
|
||||||
|
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
||||||
|
path: root,
|
||||||
|
size: fi.Size(),
|
||||||
|
mtime: fi.ModTime().Unix(),
|
||||||
|
dev: dev,
|
||||||
|
ino: ino,
|
||||||
|
}})
|
||||||
|
|
||||||
|
return nil
|
||||||
|
default:
|
||||||
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
dev, ino := inodeOfInfo(fi)
|
|
||||||
|
|
||||||
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
|
||||||
path: root,
|
|
||||||
size: fi.Size(),
|
|
||||||
mtime: fi.ModTime().Unix(),
|
|
||||||
dev: dev,
|
|
||||||
ino: ino,
|
|
||||||
}})
|
|
||||||
|
|
||||||
return nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// startWalkWorkers starts the walk worker pool. Each worker processes
|
// startWalkWorkers starts the walk worker pool. Each worker processes
|
||||||
|
|||||||
+27
-31
@@ -7,7 +7,6 @@ import (
|
|||||||
"database/sql"
|
"database/sql"
|
||||||
"encoding/hex"
|
"encoding/hex"
|
||||||
"fmt"
|
"fmt"
|
||||||
"io"
|
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"runtime"
|
"runtime"
|
||||||
@@ -400,7 +399,7 @@ func TestScanContentGate(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := dupeGroups(t, db)
|
groups := collectDupeGroups(recs)
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, same) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, same) {
|
||||||
t.Fatalf("groups = %+v, want only the identical pair %q",
|
t.Fatalf("groups = %+v, want only the identical pair %q",
|
||||||
groups, same)
|
groups, same)
|
||||||
@@ -435,7 +434,7 @@ func TestScanContentAcrossOperands(t *testing.T) {
|
|||||||
want := []string{a, b}
|
want := []string{a, b}
|
||||||
slices.Sort(want)
|
slices.Sort(want)
|
||||||
|
|
||||||
groups := dupeGroups(t, db)
|
groups := collectDupeGroups(recs)
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
||||||
}
|
}
|
||||||
@@ -462,7 +461,7 @@ func TestScanContentWithinOperand(t *testing.T) {
|
|||||||
|
|
||||||
want := []string{stored, added}
|
want := []string{stored, added}
|
||||||
|
|
||||||
groups := dupeGroups(t, db)
|
groups := collectDupeGroups(dbRecords(t, db))
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
t.Fatalf("groups = %+v, want the pair %q", groups, want)
|
||||||
}
|
}
|
||||||
@@ -520,7 +519,7 @@ func TestScanContentStalePartners(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if groups := dupeGroups(t, db); len(groups) != 0 {
|
if groups := collectDupeGroups(recs); len(groups) != 0 {
|
||||||
t.Errorf("groups = %+v, want none", groups)
|
t.Errorf("groups = %+v, want none", groups)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -571,7 +570,7 @@ func TestScanContentHashedStalePartners(t *testing.T) {
|
|||||||
|
|
||||||
// The stored records lie outside the operand and are left as they
|
// The stored records lie outside the operand and are left as they
|
||||||
// are, so they still group with each other, but not with the copy.
|
// are, so they still group with each other, but not with the copy.
|
||||||
groups := dupeGroups(t, db)
|
groups := collectDupeGroups(recs)
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, stored) {
|
||||||
t.Errorf("groups = %+v, want only the stored pair %q", groups, stored)
|
t.Errorf("groups = %+v, want only the stored pair %q", groups, stored)
|
||||||
}
|
}
|
||||||
@@ -620,7 +619,7 @@ func TestScanContentReadFailure(t *testing.T) {
|
|||||||
want := []string{a, b}
|
want := []string{a, b}
|
||||||
slices.Sort(want)
|
slices.Sort(want)
|
||||||
|
|
||||||
groups := dupeGroups(t, db)
|
groups := collectDupeGroups(dbRecords(t, db))
|
||||||
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
if len(groups) != 1 || !slices.Equal(groups[0].paths, want) {
|
||||||
t.Fatalf("groups = %+v, want the pair %q after the retry",
|
t.Fatalf("groups = %+v, want the pair %q after the retry",
|
||||||
groups, want)
|
groups, want)
|
||||||
@@ -857,11 +856,9 @@ func TestWalkFileAndSymlinkOperands(t *testing.T) {
|
|||||||
t.Fatalf("file operand: recs = %+v, errs = %d", recs, errs)
|
t.Fatalf("file operand: recs = %+v, errs = %d", recs, errs)
|
||||||
}
|
}
|
||||||
|
|
||||||
// A symlink operand that reaches the walk (it became one after
|
// A symlink operand is not followed and yields nothing.
|
||||||
// walkableRoots checked it) is not followed: it yields a warning and
|
|
||||||
// no records.
|
|
||||||
recs, errs = collectWalk(t, []string{link}, false, 2)
|
recs, errs = collectWalk(t, []string{link}, false, 2)
|
||||||
if errs != 1 || len(recs) != 0 {
|
if errs != 0 || len(recs) != 0 {
|
||||||
t.Fatalf("symlink operand: recs = %+v, errs = %d", recs, errs)
|
t.Fatalf("symlink operand: recs = %+v, errs = %d", recs, errs)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -956,16 +953,11 @@ func syncTree(t *testing.T, db *sql.DB, roots ...string) scanStats {
|
|||||||
return st
|
return st
|
||||||
}
|
}
|
||||||
|
|
||||||
// dbRecords returns every record currently in the database, in path
|
// dbRecords returns every record currently in the database.
|
||||||
// order.
|
|
||||||
func dbRecords(t *testing.T, db *sql.DB) []scanRec {
|
func dbRecords(t *testing.T, db *sql.DB) []scanRec {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
var recs []scanRec
|
recs, err := loadFileRows(t.Context(), db)
|
||||||
|
|
||||||
err := loadFileRows(t.Context(), db, func(r scanRec) {
|
|
||||||
recs = append(recs, r)
|
|
||||||
})
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
@@ -1002,10 +994,10 @@ func recordPaths(recs []scanRec) []string {
|
|||||||
|
|
||||||
// assertSmokeDupeGroups checks the file-level duplicate groups for the
|
// assertSmokeDupeGroups checks the file-level duplicate groups for the
|
||||||
// smoke tree rooted at dir.
|
// smoke tree rooted at dir.
|
||||||
func assertSmokeDupeGroups(t *testing.T, dir string, db *sql.DB) {
|
func assertSmokeDupeGroups(t *testing.T, dir string, parsed []scanRec) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
groups := dupeGroups(t, db)
|
groups := collectDupeGroups(parsed)
|
||||||
if len(groups) != 5 {
|
if len(groups) != 5 {
|
||||||
t.Fatalf("len(groups) = %d, want 5", len(groups))
|
t.Fatalf("len(groups) = %d, want 5", len(groups))
|
||||||
}
|
}
|
||||||
@@ -1030,10 +1022,11 @@ func assertSmokeDupeGroups(t *testing.T, dir string, db *sql.DB) {
|
|||||||
|
|
||||||
// assertSmokeTreeGroups checks the duplicate-tree groups for the smoke
|
// assertSmokeTreeGroups checks the duplicate-tree groups for the smoke
|
||||||
// tree rooted at dir.
|
// tree rooted at dir.
|
||||||
func assertSmokeTreeGroups(t *testing.T, dir string, db *sql.DB) {
|
func assertSmokeTreeGroups(t *testing.T, dir string, parsed []scanRec) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
super, dirs := dbTree(t, db)
|
super, dirs := buildHierarchy(parsed)
|
||||||
|
super.compute()
|
||||||
|
|
||||||
tg := collectTreeGroups(dirs, super)
|
tg := collectTreeGroups(dirs, super)
|
||||||
if len(tg) != 1 {
|
if len(tg) != 1 {
|
||||||
@@ -1067,8 +1060,8 @@ func TestScanPipeline(t *testing.T) {
|
|||||||
t.Fatalf("len(records) = %d, want %d", len(parsed), smokeTreeFiles)
|
t.Fatalf("len(records) = %d, want %d", len(parsed), smokeTreeFiles)
|
||||||
}
|
}
|
||||||
|
|
||||||
assertSmokeDupeGroups(t, dir, db)
|
assertSmokeDupeGroups(t, dir, parsed)
|
||||||
assertSmokeTreeGroups(t, dir, db)
|
assertSmokeTreeGroups(t, dir, parsed)
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestSyncScanUnchangedReuse(t *testing.T) {
|
func TestSyncScanUnchangedReuse(t *testing.T) {
|
||||||
@@ -1302,7 +1295,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if groups := dupeGroups(t, db); len(groups) != 0 {
|
if groups := collectDupeGroups(recs); len(groups) != 0 {
|
||||||
t.Fatalf("groups = %+v, want none from unhashed records", groups)
|
t.Fatalf("groups = %+v, want none from unhashed records", groups)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1317,7 +1310,7 @@ func TestScanSkipsUniqueSizes(t *testing.T) {
|
|||||||
st)
|
st)
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := dupeGroups(t, db)
|
groups := collectDupeGroups(dbRecords(t, db))
|
||||||
if len(groups) != 1 {
|
if len(groups) != 1 {
|
||||||
t.Fatalf("groups = %+v, want the a/c pair", groups)
|
t.Fatalf("groups = %+v, want the a/c pair", groups)
|
||||||
}
|
}
|
||||||
@@ -1353,7 +1346,8 @@ func TestTreesUnhashedNeverEqual(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
for name, unknown := range cases {
|
for name, unknown := range cases {
|
||||||
super, dirs := treeOf(t, append(slices.Clone(shared), unknown...))
|
super, dirs := buildHierarchy(append(slices.Clone(shared), unknown...))
|
||||||
|
super.compute()
|
||||||
|
|
||||||
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
|
if tg := collectTreeGroups(dirs, super); len(tg) != 0 {
|
||||||
t.Errorf("%s: tree groups = %d, want 0 (the files may differ)",
|
t.Errorf("%s: tree groups = %d, want 0 (the files may differ)",
|
||||||
@@ -1390,7 +1384,7 @@ func TestScanHardlinksReadOnce(t *testing.T) {
|
|||||||
t.Fatalf("hardlink hashes differ: %+v vs %+v", ra, rb)
|
t.Fatalf("hardlink hashes differ: %+v vs %+v", ra, rb)
|
||||||
}
|
}
|
||||||
|
|
||||||
if groups := dupeGroups(t, db); len(groups) != 1 {
|
if groups := collectDupeGroups(recs); len(groups) != 1 {
|
||||||
t.Fatalf("groups = %+v, want the hardlink pair", groups)
|
t.Fatalf("groups = %+v, want the hardlink pair", groups)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1554,7 +1548,7 @@ func TestScanHashWriteFailureUnwindsPool(t *testing.T) {
|
|||||||
|
|
||||||
code := run([]string{
|
code := run([]string{
|
||||||
cmdScan, "--workers", strconv.Itoa(hashLeakWorkers), dir,
|
cmdScan, "--workers", strconv.Itoa(hashLeakWorkers), dir,
|
||||||
}, io.Discard, &stderr)
|
}, &stderr)
|
||||||
if code != exitFatal {
|
if code != exitFatal {
|
||||||
t.Fatalf("run(scan) = %d, want %d; stderr: %s",
|
t.Fatalf("run(scan) = %d, want %d; stderr: %s",
|
||||||
code, exitFatal, stderr.String())
|
code, exitFatal, stderr.String())
|
||||||
@@ -1631,8 +1625,10 @@ func TestReportsNeverTouchFilesystem(t *testing.T) {
|
|||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
|
|
||||||
assertSmokeDupeGroups(t, dir, db)
|
recs := dbRecords(t, db)
|
||||||
assertSmokeTreeGroups(t, dir, db)
|
|
||||||
|
assertSmokeDupeGroups(t, dir, recs)
|
||||||
|
assertSmokeTreeGroups(t, dir, recs)
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestUnderRoot(t *testing.T) {
|
func TestUnderRoot(t *testing.T) {
|
||||||
|
|||||||
+4
-4
@@ -16,10 +16,10 @@
|
|||||||
# implies the repo is green.
|
# implies the repo is green.
|
||||||
#
|
#
|
||||||
# That implication holds only because of CHECK_EPOCH. A COPY layer is
|
# That implication holds only because of CHECK_EPOCH. A COPY layer is
|
||||||
# invalidated only by changed content, and a rebuild of an unchanged
|
# invalidated by changed content, and a merge commit's tree is
|
||||||
# checkout sends the same content, so without a fresh value here Docker
|
# byte-identical to the branch head it merges, so without a fresh value
|
||||||
# serves the gate layers from cache and the build reports a green it
|
# here Docker serves the gate layers from cache and the build reports a
|
||||||
# never earned. Passing the current epoch invalidates the gate
|
# green it never earned. Passing the current epoch invalidates the gate
|
||||||
# layers on every run while leaving the pinned base images and
|
# layers on every run while leaving the pinned base images and
|
||||||
# go mod download cached; see the Dockerfile for the placement.
|
# go mod download cached; see the Dockerfile for the placement.
|
||||||
set -eu
|
set -eu
|
||||||
|
|||||||
@@ -5,59 +5,49 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"crypto/sha256"
|
"crypto/sha256"
|
||||||
"fmt"
|
"fmt"
|
||||||
"io"
|
|
||||||
"os"
|
"os"
|
||||||
"slices"
|
"slices"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
|
|
||||||
// treeNode is one directory reconstructed from the record paths.
|
// fileSig is a file's duplicate signature; mtime is excluded.
|
||||||
|
type fileSig struct {
|
||||||
|
size int64
|
||||||
|
head string
|
||||||
|
tail string
|
||||||
|
content string
|
||||||
|
}
|
||||||
|
|
||||||
|
// treeNode is one directory reconstructed from the scan stream.
|
||||||
type treeNode struct {
|
type treeNode struct {
|
||||||
path string
|
path string
|
||||||
parent *treeNode
|
parent *treeNode
|
||||||
// entries holds the serialized child entries until the digest is
|
dirs map[string]*treeNode
|
||||||
// computed from them, and is then dropped.
|
files map[string]fileSig
|
||||||
entries []string
|
|
||||||
digest [sha256.Size]byte
|
digest [sha256.Size]byte
|
||||||
fileCount int64
|
fileCount int64
|
||||||
totalSize int64
|
totalSize int64
|
||||||
}
|
}
|
||||||
|
|
||||||
// runTrees implements the trees subcommand: it reads every record from
|
// runTrees implements the trees subcommand: it reads every record from
|
||||||
// the database in path order, reconstructs the directory hierarchy from
|
// the database, reconstructs the directory hierarchy from the record
|
||||||
// the record paths, computes a Merkle-style digest per directory, and
|
// paths, computes a Merkle-style digest per directory, and prints
|
||||||
// prints maximal duplicate-tree groups as TSV on stdout. It never
|
// maximal duplicate-tree groups as TSV on stdout. It never touches the
|
||||||
// touches the scanned filesystem; its only I/O is the database, stdout,
|
// scanned filesystem; its only I/O is the database, stdout, and
|
||||||
// and stderr. Any database problem, including a missing database, is
|
// stderr.
|
||||||
// fatal.
|
func runTrees(ctx context.Context) error {
|
||||||
func runTrees(ctx context.Context, stdout io.Writer) error {
|
recs, err := loadRecords(ctx)
|
||||||
dbPath := databasePath()
|
|
||||||
|
|
||||||
db, err := openReportDatabase(ctx, dbPath)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
defer func() { _ = db.Close() }()
|
super, allDirs := buildHierarchy(recs)
|
||||||
|
super.compute()
|
||||||
records := 0
|
|
||||||
tree := newTreeBuilder()
|
|
||||||
|
|
||||||
err = loadFileRows(ctx, db, func(r scanRec) {
|
|
||||||
records++
|
|
||||||
|
|
||||||
tree.add(r)
|
|
||||||
})
|
|
||||||
if err != nil {
|
|
||||||
return fmt.Errorf("database %s: %w", dbPath, err)
|
|
||||||
}
|
|
||||||
|
|
||||||
super, allDirs := tree.finish()
|
|
||||||
|
|
||||||
dupes := collectTreeGroups(allDirs, super)
|
dupes := collectTreeGroups(allDirs, super)
|
||||||
|
|
||||||
out := bufio.NewWriterSize(stdout, ioBufSize)
|
out := bufio.NewWriterSize(os.Stdout, ioBufSize)
|
||||||
|
|
||||||
_, err = fmt.Fprintln(out, "first\tdupe\tfiles\tsize")
|
_, err = fmt.Fprintln(out, "first\tdupe\tfiles\tsize")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -72,8 +62,7 @@ func runTrees(ctx context.Context, stdout io.Writer) error {
|
|||||||
first := g[0]
|
first := g[0]
|
||||||
for _, n := range g[1:] {
|
for _, n := range g[1:] {
|
||||||
_, err = fmt.Fprintf(out, "%s\t%s\t%d\t%d\n",
|
_, err = fmt.Fprintf(out, "%s\t%s\t%d\t%d\n",
|
||||||
escapePath(first.path), escapePath(n.path),
|
first.path, n.path, first.fileCount, first.totalSize)
|
||||||
first.fileCount, first.totalSize)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("write stdout: %w", err)
|
return fmt.Errorf("write stdout: %w", err)
|
||||||
}
|
}
|
||||||
@@ -91,128 +80,65 @@ func runTrees(ctx context.Context, stdout io.Writer) error {
|
|||||||
fmt.Fprintf(os.Stderr,
|
fmt.Fprintf(os.Stderr,
|
||||||
"trees: %d records read, %d duplicate tree groups, %d dupe trees, "+
|
"trees: %d records read, %d duplicate tree groups, %d dupe trees, "+
|
||||||
"%s reclaimable\n",
|
"%s reclaimable\n",
|
||||||
records, len(dupes), dupeTrees, humanBytes(reclaimable))
|
len(recs), len(dupes), dupeTrees, humanBytes(reclaimable))
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// treeBuilder reconstructs the directory hierarchy from records added
|
// buildHierarchy reconstructs the directory hierarchy from the record
|
||||||
// in path order, under a synthetic super-root. Paths are split on "/";
|
// paths under a synthetic super-root. Paths are split on "/"; for
|
||||||
// for absolute paths the first component is empty, which becomes the
|
// absolute paths the first component is empty, which simply becomes a
|
||||||
// top-level directory with path "/". In path order all the paths under
|
// top-level node representing "/". It returns the super-root and every
|
||||||
// one directory come together, so a directory is complete once a path
|
// directory node created.
|
||||||
// outside it is added: its digest is computed then and its entries are
|
func buildHierarchy(recs []scanRec) (*treeNode, []*treeNode) {
|
||||||
// dropped. Only the directories holding the latest path keep entries.
|
|
||||||
type treeBuilder struct {
|
|
||||||
super *treeNode
|
|
||||||
// open lists the directories holding the latest path, outermost
|
|
||||||
// first, starting with the super-root; names[i] is open[i]'s name.
|
|
||||||
open []*treeNode
|
|
||||||
names []string
|
|
||||||
// dirs lists every completed directory.
|
|
||||||
dirs []*treeNode
|
|
||||||
}
|
|
||||||
|
|
||||||
func newTreeBuilder() *treeBuilder {
|
|
||||||
super := &treeNode{}
|
super := &treeNode{}
|
||||||
|
|
||||||
return &treeBuilder{
|
var allDirs []*treeNode
|
||||||
super: super,
|
|
||||||
open: []*treeNode{super},
|
|
||||||
names: []string{""},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// add adds one record. Each record must come after the previous one in
|
for _, r := range recs {
|
||||||
// path order (byte order); otherwise a completed directory would be
|
comps := strings.Split(r.path, "/")
|
||||||
// started again as a second directory with the same path.
|
|
||||||
func (b *treeBuilder) add(r scanRec) {
|
|
||||||
comps := strings.Split(r.path, "/")
|
|
||||||
dirNames, name := comps[:len(comps)-1], comps[len(comps)-1]
|
|
||||||
|
|
||||||
// Keep the open directories that hold this path; complete the rest.
|
node := super
|
||||||
depth := 1
|
for _, c := range comps[:len(comps)-1] {
|
||||||
for depth < len(b.open) && depth <= len(dirNames) &&
|
child := node.dirs[c]
|
||||||
b.names[depth] == dirNames[depth-1] {
|
if child == nil {
|
||||||
depth++
|
childPath := c
|
||||||
|
if node != super {
|
||||||
|
childPath = node.path + "/" + c
|
||||||
|
}
|
||||||
|
|
||||||
|
child = &treeNode{path: childPath, parent: node}
|
||||||
|
if node.dirs == nil {
|
||||||
|
node.dirs = make(map[string]*treeNode)
|
||||||
|
}
|
||||||
|
|
||||||
|
node.dirs[c] = child
|
||||||
|
allDirs = append(allDirs, child)
|
||||||
|
}
|
||||||
|
|
||||||
|
node = child
|
||||||
|
}
|
||||||
|
|
||||||
|
if node.files == nil {
|
||||||
|
node.files = make(map[string]fileSig)
|
||||||
|
}
|
||||||
|
|
||||||
|
sig := fileSig{
|
||||||
|
size: r.size, head: r.head, tail: r.tail, content: r.content,
|
||||||
|
}
|
||||||
|
|
||||||
|
// A record without a content hash has unknown content (README
|
||||||
|
// "Database"): give it a signature no other file can share, so
|
||||||
|
// trees containing it never compare equal. Real hashes are
|
||||||
|
// hex, so the NUL-prefixed form cannot collide.
|
||||||
|
if sig.content == "" {
|
||||||
|
sig.content = "unhashed\x00" + r.path
|
||||||
|
}
|
||||||
|
|
||||||
|
node.files[comps[len(comps)-1]] = sig
|
||||||
}
|
}
|
||||||
|
|
||||||
b.closeTo(depth)
|
return super, allDirs
|
||||||
|
|
||||||
for _, c := range dirNames[depth-1:] {
|
|
||||||
b.openDir(c)
|
|
||||||
}
|
|
||||||
|
|
||||||
dir := b.open[len(b.open)-1]
|
|
||||||
dir.entries = append(dir.entries, fileEntry(name, r))
|
|
||||||
dir.fileCount++
|
|
||||||
dir.totalSize += r.size
|
|
||||||
}
|
|
||||||
|
|
||||||
// openDir starts the directory called name inside the innermost open
|
|
||||||
// one.
|
|
||||||
func (b *treeBuilder) openDir(name string) {
|
|
||||||
parent := b.open[len(b.open)-1]
|
|
||||||
path := parent.path + "/" + name
|
|
||||||
|
|
||||||
// The root directory's path is "/", not empty, and its children's
|
|
||||||
// paths start with one slash, not two.
|
|
||||||
switch {
|
|
||||||
case parent == b.super && name == "":
|
|
||||||
path = "/"
|
|
||||||
case parent == b.super:
|
|
||||||
path = name
|
|
||||||
case parent.path == "/":
|
|
||||||
path = "/" + name
|
|
||||||
}
|
|
||||||
|
|
||||||
b.open = append(b.open, &treeNode{path: path, parent: parent})
|
|
||||||
b.names = append(b.names, name)
|
|
||||||
}
|
|
||||||
|
|
||||||
// closeTo completes the open directories after the first n, innermost
|
|
||||||
// first: each one's digest is computed and entered in its parent along
|
|
||||||
// with its totals.
|
|
||||||
func (b *treeBuilder) closeTo(n int) {
|
|
||||||
for len(b.open) > n {
|
|
||||||
last := len(b.open) - 1
|
|
||||||
dir, name := b.open[last], b.names[last]
|
|
||||||
b.open, b.names = b.open[:last], b.names[:last]
|
|
||||||
|
|
||||||
dir.computeDigest()
|
|
||||||
|
|
||||||
dir.parent.entries = append(dir.parent.entries,
|
|
||||||
"d\x00"+name+"\x00"+string(dir.digest[:]))
|
|
||||||
dir.parent.fileCount += dir.fileCount
|
|
||||||
dir.parent.totalSize += dir.totalSize
|
|
||||||
b.dirs = append(b.dirs, dir)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// finish completes every open directory and returns the super-root and
|
|
||||||
// every directory.
|
|
||||||
func (b *treeBuilder) finish() (*treeNode, []*treeNode) {
|
|
||||||
b.closeTo(1)
|
|
||||||
|
|
||||||
return b.super, b.dirs
|
|
||||||
}
|
|
||||||
|
|
||||||
// fileEntry serializes a file child for its directory's digest: its
|
|
||||||
// name and its signature (size, head, tail, content); mtime is
|
|
||||||
// excluded.
|
|
||||||
func fileEntry(name string, r scanRec) string {
|
|
||||||
content := r.content
|
|
||||||
|
|
||||||
// A record without a content hash has unknown content (README
|
|
||||||
// "Database"): give it a signature no other file can share, so
|
|
||||||
// trees containing it never compare equal. Real hashes are hex, so
|
|
||||||
// the NUL-prefixed form cannot collide.
|
|
||||||
if content == "" {
|
|
||||||
content = "unhashed\x00" + r.path
|
|
||||||
}
|
|
||||||
|
|
||||||
return "f\x00" + name + "\x00" + strconv.FormatInt(r.size, 10) +
|
|
||||||
"\x00" + r.head + "\x00" + r.tail + "\x00" + content
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// collectTreeGroups groups directories by digest and returns every
|
// collectTreeGroups groups directories by digest and returns every
|
||||||
@@ -254,22 +180,38 @@ func collectTreeGroups(allDirs []*treeNode, super *treeNode) [][]*treeNode {
|
|||||||
return dupes
|
return dupes
|
||||||
}
|
}
|
||||||
|
|
||||||
// computeDigest sets n's digest and drops its entries. A directory's
|
// compute fills in digest, fileCount, and totalSize for n and all of
|
||||||
// digest is the SHA-256 of its child entries — files serialized with
|
// its descendants. A directory's digest is the SHA-256 of its child
|
||||||
// name and signature, subdirectories with name and recursive digest —
|
// entries — files serialized with name and signature, subdirectories
|
||||||
// sorted byte-lexicographically. Filenames cannot contain NUL or "/",
|
// with name and recursive digest — sorted byte-lexicographically.
|
||||||
// so NUL delimiters are unambiguous.
|
// Filenames cannot contain NUL or "/", so NUL delimiters are
|
||||||
func (n *treeNode) computeDigest() {
|
// unambiguous.
|
||||||
slices.Sort(n.entries)
|
func (n *treeNode) compute() {
|
||||||
|
entries := make([]string, 0, len(n.dirs)+len(n.files))
|
||||||
|
for name, sig := range n.files {
|
||||||
|
entries = append(entries,
|
||||||
|
"f\x00"+name+"\x00"+strconv.FormatInt(sig.size, 10)+
|
||||||
|
"\x00"+sig.head+"\x00"+sig.tail+"\x00"+sig.content)
|
||||||
|
n.fileCount++
|
||||||
|
n.totalSize += sig.size
|
||||||
|
}
|
||||||
|
|
||||||
|
for name, child := range n.dirs {
|
||||||
|
child.compute()
|
||||||
|
entries = append(entries, "d\x00"+name+"\x00"+string(child.digest[:]))
|
||||||
|
n.fileCount += child.fileCount
|
||||||
|
n.totalSize += child.totalSize
|
||||||
|
}
|
||||||
|
|
||||||
|
slices.Sort(entries)
|
||||||
|
|
||||||
h := sha256.New()
|
h := sha256.New()
|
||||||
for _, e := range n.entries {
|
for _, e := range entries {
|
||||||
h.Write([]byte(e))
|
h.Write([]byte(e))
|
||||||
h.Write([]byte{0})
|
h.Write([]byte{0})
|
||||||
}
|
}
|
||||||
|
|
||||||
copy(n.digest[:], h.Sum(nil))
|
copy(n.digest[:], h.Sum(nil))
|
||||||
n.entries = nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// suppressed reports whether a duplicate-tree group is non-maximal: its
|
// suppressed reports whether a duplicate-tree group is non-maximal: its
|
||||||
|
|||||||
+17
-129
@@ -1,8 +1,6 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
|
||||||
"database/sql"
|
|
||||||
"slices"
|
"slices"
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
@@ -31,36 +29,6 @@ func smokeTreeRecs() []scanRec {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// dbTree builds the directory hierarchy from the records in db the way
|
|
||||||
// trees does, and returns the super-root and every directory.
|
|
||||||
func dbTree(t *testing.T, db *sql.DB) (*treeNode, []*treeNode) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
tree := newTreeBuilder()
|
|
||||||
|
|
||||||
err := loadFileRows(t.Context(), db, tree.add)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return tree.finish()
|
|
||||||
}
|
|
||||||
|
|
||||||
// treeOf writes recs into a fresh database and builds the directory
|
|
||||||
// hierarchy from it the way trees does.
|
|
||||||
func treeOf(t *testing.T, recs []scanRec) (*treeNode, []*treeNode) {
|
|
||||||
t.Helper()
|
|
||||||
|
|
||||||
db := openTestDB(t)
|
|
||||||
|
|
||||||
err := applyChanges(t.Context(), db, recs, nil, nil)
|
|
||||||
if err != nil {
|
|
||||||
t.Fatal(err)
|
|
||||||
}
|
|
||||||
|
|
||||||
return dbTree(t, db)
|
|
||||||
}
|
|
||||||
|
|
||||||
// nodeByPath finds the directory node with the given path.
|
// nodeByPath finds the directory node with the given path.
|
||||||
func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode {
|
func nodeByPath(t *testing.T, dirs []*treeNode, path string) *treeNode {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
@@ -91,10 +59,11 @@ func groupPaths(groups [][]*treeNode) [][]string {
|
|||||||
return out
|
return out
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestTreeCounts(t *testing.T) {
|
func TestBuildHierarchyCounts(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
_, dirs := treeOf(t, smokeTreeRecs())
|
super, dirs := buildHierarchy(smokeTreeRecs())
|
||||||
|
super.compute()
|
||||||
|
|
||||||
d := nodeByPath(t, dirs, "/d")
|
d := nodeByPath(t, dirs, "/d")
|
||||||
if d.fileCount != 6 || d.totalSize != 9300 {
|
if d.fileCount != 6 || d.totalSize != 9300 {
|
||||||
@@ -115,98 +84,11 @@ func TestTreeCounts(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestTreeRootPath(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
// The root directory's path is "/", never empty, and its
|
|
||||||
// children's paths start with a single slash.
|
|
||||||
_, dirs := treeOf(t, []scanRec{{path: "/f"}, {path: "/srv/g"}})
|
|
||||||
|
|
||||||
got := make([]string, 0, len(dirs))
|
|
||||||
for _, d := range dirs {
|
|
||||||
got = append(got, d.path)
|
|
||||||
}
|
|
||||||
|
|
||||||
slices.Sort(got)
|
|
||||||
|
|
||||||
want := []string{"/", "/srv"}
|
|
||||||
if !slices.Equal(got, want) {
|
|
||||||
t.Fatalf("directory paths = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestTreeNamesSortingBeforeSlash(t *testing.T) {
|
|
||||||
t.Parallel()
|
|
||||||
|
|
||||||
// In path order "/a/b-x/f" and "/a/b.txt" come between the file
|
|
||||||
// "/a/b" and "/a/b/f", because "-" and "." sort before "/". Each
|
|
||||||
// directory must still be built once, whole, so /a matches /c.
|
|
||||||
recs := make([]scanRec, 0, 8)
|
|
||||||
|
|
||||||
for _, top := range []string{"/a", "/c"} {
|
|
||||||
for _, p := range []string{"/b", "/b-x/f", "/b.txt", "/b/f"} {
|
|
||||||
content := "c"
|
|
||||||
if p == "/b-x/f" {
|
|
||||||
content = "other"
|
|
||||||
}
|
|
||||||
|
|
||||||
recs = append(recs, scanRec{
|
|
||||||
size: 1, head: "h", tail: "t", content: content, path: top + p,
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
super, dirs := treeOf(t, recs)
|
|
||||||
|
|
||||||
got := make([]string, 0, len(dirs))
|
|
||||||
for _, d := range dirs {
|
|
||||||
got = append(got, d.path)
|
|
||||||
}
|
|
||||||
|
|
||||||
slices.Sort(got)
|
|
||||||
|
|
||||||
want := []string{"/", "/a", "/a/b", "/a/b-x", "/c", "/c/b", "/c/b-x"}
|
|
||||||
if !slices.Equal(got, want) {
|
|
||||||
t.Fatalf("directory paths = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
|
|
||||||
groups := collectTreeGroups(dirs, super)
|
|
||||||
|
|
||||||
gotGroups := groupPaths(groups)
|
|
||||||
wantGroups := [][]string{{"/a", "/c"}}
|
|
||||||
|
|
||||||
if !slices.EqualFunc(gotGroups, wantGroups, slices.Equal) {
|
|
||||||
t.Fatalf("groups = %v, want %v", gotGroups, wantGroups)
|
|
||||||
}
|
|
||||||
|
|
||||||
if groups[0][0].fileCount != 4 || groups[0][0].totalSize != 4 {
|
|
||||||
t.Errorf("group totals: %d files %d bytes, want 4 4",
|
|
||||||
groups[0][0].fileCount, groups[0][0].totalSize)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestRunTreesEscapesPaths(t *testing.T) {
|
|
||||||
t.Setenv(databaseEnv, seedDatabase(t, awkwardPairRecs()))
|
|
||||||
|
|
||||||
var stdout, stderr bytes.Buffer
|
|
||||||
|
|
||||||
code := run([]string{cmdTrees}, &stdout, &stderr)
|
|
||||||
if code != exitOK {
|
|
||||||
t.Fatalf("run(trees) = %d, want %d; stderr: %s",
|
|
||||||
code, exitOK, stderr.String())
|
|
||||||
}
|
|
||||||
|
|
||||||
want := "first\tdupe\tfiles\tsize\n" +
|
|
||||||
`/d/\tone\ntwo\rthree\\four` + "\t/d/A\t1\t5\n"
|
|
||||||
if got := stdout.String(); got != want {
|
|
||||||
t.Errorf("stdout = %q, want %q", got, want)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func TestTreeDigests(t *testing.T) {
|
func TestTreeDigests(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
_, dirs := treeOf(t, smokeTreeRecs())
|
super, dirs := buildHierarchy(smokeTreeRecs())
|
||||||
|
super.compute()
|
||||||
|
|
||||||
t1 := nodeByPath(t, dirs, "/d/t1")
|
t1 := nodeByPath(t, dirs, "/d/t1")
|
||||||
t2 := nodeByPath(t, dirs, "/d/t2")
|
t2 := nodeByPath(t, dirs, "/d/t2")
|
||||||
@@ -239,7 +121,8 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
|
|||||||
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
|
{size: 10, head: "DIFF", tail: sharedTail, content: "c", path: "/r/b/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
_, dirs := treeOf(t, recs)
|
super, dirs := buildHierarchy(recs)
|
||||||
|
super.compute()
|
||||||
|
|
||||||
a := nodeByPath(t, dirs, "/r/a")
|
a := nodeByPath(t, dirs, "/r/a")
|
||||||
b := nodeByPath(t, dirs, "/r/b")
|
b := nodeByPath(t, dirs, "/r/b")
|
||||||
@@ -252,7 +135,8 @@ func TestTreeDigestContentSensitivity(t *testing.T) {
|
|||||||
func TestCollectTreeGroupsMaximal(t *testing.T) {
|
func TestCollectTreeGroupsMaximal(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
super, dirs := treeOf(t, smokeTreeRecs())
|
super, dirs := buildHierarchy(smokeTreeRecs())
|
||||||
|
super.compute()
|
||||||
|
|
||||||
groups := collectTreeGroups(dirs, super)
|
groups := collectTreeGroups(dirs, super)
|
||||||
|
|
||||||
@@ -276,14 +160,16 @@ func TestCollectTreeGroupsDeterministic(t *testing.T) {
|
|||||||
|
|
||||||
recs := smokeTreeRecs()
|
recs := smokeTreeRecs()
|
||||||
|
|
||||||
super, dirs := treeOf(t, recs)
|
super, dirs := buildHierarchy(recs)
|
||||||
|
super.compute()
|
||||||
|
|
||||||
forward := groupPaths(collectTreeGroups(dirs, super))
|
forward := groupPaths(collectTreeGroups(dirs, super))
|
||||||
|
|
||||||
reversed := slices.Clone(recs)
|
reversed := slices.Clone(recs)
|
||||||
slices.Reverse(reversed)
|
slices.Reverse(reversed)
|
||||||
|
|
||||||
superR, dirsR := treeOf(t, reversed)
|
superR, dirsR := buildHierarchy(reversed)
|
||||||
|
superR.compute()
|
||||||
|
|
||||||
backward := groupPaths(collectTreeGroups(dirsR, superR))
|
backward := groupPaths(collectTreeGroups(dirsR, superR))
|
||||||
if !slices.EqualFunc(forward, backward, slices.Equal) {
|
if !slices.EqualFunc(forward, backward, slices.Equal) {
|
||||||
@@ -302,7 +188,8 @@ func TestCollectTreeGroupsSiblings(t *testing.T) {
|
|||||||
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
|
{size: 10, head: "h", tail: "t", content: "c", path: "/p/x2/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := treeOf(t, recs)
|
super, dirs := buildHierarchy(recs)
|
||||||
|
super.compute()
|
||||||
|
|
||||||
got := groupPaths(collectTreeGroups(dirs, super))
|
got := groupPaths(collectTreeGroups(dirs, super))
|
||||||
|
|
||||||
@@ -324,7 +211,8 @@ func TestCollectTreeGroupsDifferingParents(t *testing.T) {
|
|||||||
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
|
{size: 10, head: "h", tail: "t", content: "c", path: "/q/b/x/f"},
|
||||||
}
|
}
|
||||||
|
|
||||||
super, dirs := treeOf(t, recs)
|
super, dirs := buildHierarchy(recs)
|
||||||
|
super.compute()
|
||||||
|
|
||||||
got := groupPaths(collectTreeGroups(dirs, super))
|
got := groupPaths(collectTreeGroups(dirs, super))
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user