diff --git a/.prettierignore b/.prettierignore new file mode 100644 index 0000000..23d67fc --- /dev/null +++ b/.prettierignore @@ -0,0 +1,2 @@ +node_modules/ +yarn.lock diff --git a/.prettierrc b/.prettierrc new file mode 100644 index 0000000..8af31cd --- /dev/null +++ b/.prettierrc @@ -0,0 +1,4 @@ +{ + "tabWidth": 4, + "proseWrap": "always" +} diff --git a/Dockerfile b/Dockerfile index 98a2174..b911abe 100644 --- a/Dockerfile +++ b/Dockerfile @@ -12,9 +12,9 @@ COPY . . # build would exit 0 having run nothing. script/cibuild and # script/docker pass a fresh CHECK_EPOCH on every invocation. # -# Two properties this depends on. ARG is per-stage, so the build stage -# below declares it again; one declaration here would leave that -# stage's gate cacheable. And each gate RUN must reference the value, +# Two properties this depends on. ARG is per-stage, so the markdown and +# build stages below declare it again; one declaration here would leave +# their gates cacheable. And each gate RUN must reference the value, # because BuildKit hashes the expanded command: a declared but # unreferenced ARG invalidates nothing. # @@ -27,9 +27,16 @@ ARG CHECK_EPOCH # target now runs `docker build -f Dockerfile.lint`, and a docker build # cannot run a docker build: routing the gate through make would mean # nesting docker inside this image. Same reason `make check` is gone -# from the build stage below. `make fmt-check` stays as it is — it is a -# gate, not the aggregate, and it shells out to nothing. -RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check +# from the build stage below, and `make fmt-check` from both stages: it +# runs prettier through docker too. Its gofmt half is the step below, +# its Markdown half the markdown stage further down. gofmt's output is +# assigned to a variable first so that its own exit status, as when it +# cannot parse a file, still fails the step. +RUN echo "gate gofmt, epoch ${CHECK_EPOCH}" && \ + files="$(gofmt -s -l .)" && \ + if [ -n "$files" ]; then \ + echo "gofmt: files not formatted:" >&2; echo "$files" >&2; exit 1; \ + fi # The FROM above and the one in Dockerfile.lint pin the same linter # twice, and nothing else keeps them in sync; this fails the build when @@ -46,6 +53,30 @@ RUN echo "gate config verify, epoch ${CHECK_EPOCH}" && \ RUN echo "gate lint, epoch ${CHECK_EPOCH}" && \ golangci-lint run --config .golangci.yml ./... +# Prettier stage: the prettier that formats this repository's Markdown, +# never installed on a host. script/fmt and script/fmt-check build this +# stage alone and run it with the repository mounted on /src. prettier +# is installed in /tools so that the repository, mounted or copied onto +# /src, cannot hide it. +# node:22-alpine, 2026-02-22 +FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS prettier +WORKDIR /tools +# yarn.lock pins prettier by hash, and --frozen-lockfile fails rather +# than install anything yarn.lock does not name. +COPY package.json yarn.lock ./ +RUN yarn install --frozen-lockfile +ENV PATH=/tools/node_modules/.bin:$PATH +WORKDIR /src + +# Markdown stage: the Markdown half of `make fmt-check`, as a gate. +FROM prettier AS markdown +COPY . . +# Second per-stage declaration of the gate cache-buster; see the lint +# stage above. +ARG CHECK_EPOCH +RUN echo "gate prettier, epoch ${CHECK_EPOCH}" && \ + prettier --check '**/*.md' --tab-width 4 --prose-wrap always + # Build stage # golang:1.25-alpine, 2026-07-23 FROM golang@sha256:56961d79ea8129efddcc0b8643fd8a5416b4e6228cfd477e3fd61deb2672c587 AS builder @@ -70,15 +101,17 @@ ENV GOCACHE=/home/builder/.cache/go-build WORKDIR /src -# No-op file copy whose only purpose is the build-graph edge: it is what -# makes this stage depend on the lint stage, and so what forces BuildKit -# to finish fmt-check, the pin guard and lint before compilation and -# tests start. Remove it and the fail-fast design dies silently — the -# build stops gating on lint and still exits 0. It replaces a copy of -# the linter binary itself, which is no longer wanted here: nothing in -# this stage runs the linter, because `make lint` is now a docker build -# and a docker build cannot run inside one. +# No-op file copies whose only purpose is the build-graph edge: they are +# what make this stage depend on the lint and markdown stages, and so +# what forces BuildKit to finish gofmt, the pin guard, lint and prettier +# before compilation and tests start. Remove one and the fail-fast +# design dies silently — the build stops gating on that stage and still +# exits 0. The first replaces a copy of the linter binary itself, which +# is no longer wanted here: nothing in this stage runs the linter, +# because `make lint` is now a docker build and a docker build cannot +# run inside one. COPY --from=lint /src/go.sum /dev/null +COPY --from=markdown /src/go.sum /dev/null # Install development prerequisites the same way a developer does, # rather than duplicating the installs inline. Only script/ and the @@ -112,19 +145,19 @@ USER builder # permission-denied test paths are exercised legitimately (root would # bypass the chmod(0) the tests rely on). # -# The gates are the individual targets, not `make check`: that aggregate -# runs `script/lint`, which is now a docker build, and nothing inside an -# image build may shell out to docker. Lint is not skipped by this — it -# ran in the lint stage above, which this stage's COPY --from makes a -# prerequisite. `make`, not the scripts directly, because the Makefile's +# The gate is `make test`, not `make check`: that aggregate runs +# `script/lint` and `script/fmt-check`, which both run docker, and +# nothing inside an image build may shell out to docker. Lint and the +# format checks are not skipped by this — they ran in the lint and +# markdown stages above, which this stage's COPY --from lines make +# prerequisites. `make`, not the script directly, because the Makefile's # `export CGO_ENABLED = 0` applies only to what it invokes. # -# Second per-stage declaration of the gate cache-buster; see the lint +# Third per-stage declaration of the gate cache-buster; see the lint # stage above for why one is not enough. It is placed after USER so the # drop to the unprivileged user still happens before the checks run. ARG CHECK_EPOCH RUN echo "gate test, epoch ${CHECK_EPOCH}" && make test -RUN echo "gate fmt-check, epoch ${CHECK_EPOCH}" && make fmt-check # The version stamped into the binary: the VERSION build argument when # one is given, otherwise `git describe --tags --always` of the .git in diff --git a/README.md b/README.md index 4e4ac42..8db99a1 100644 --- a/README.md +++ b/README.md @@ -2,25 +2,23 @@ ## Description -`sfdupes` is an MIT-licensed Go CLI tool by -[@sneak](https://sneak.berlin) that quickly identifies *candidate* -duplicate files — and, ultimately, entire duplicate directory trees — -across very large filesystems without reading every byte of every file. -Files are considered duplicates when their sizes are equal and they -agree on a short ladder of hashes. A file under 10 MiB is hashed in full -and compared directly. A larger file is gated first on the SHA-256 of -its first 64 KiB and of its last 64 KiB, and only when its size and -both of those match another file's is it read for a content hash to -compare — the SHA-256 of the whole file when it is under 50 MiB, or of -gigabyte-spaced 1 MiB samples when it is 50 MiB or larger. Below 50 MiB -the content hash is proof of identical content; at or above 50 MiB it is -a strong candidate signal rather than proof, because the gaps between -samples are never read. The intended use is -finding duplicate downloads and duplicated directory trees on -multi-terabyte ZFS servers where reading every byte of every file is -prohibitively expensive. `scan` maintains a persistent SQLite database of file -signatures that survives between runs, so it can be run from cron and -the reports can be generated at any time from the most recent scan. +`sfdupes` is an MIT-licensed Go CLI tool by [@sneak](https://sneak.berlin) that +quickly identifies _candidate_ duplicate files — and, ultimately, entire +duplicate directory trees — across very large filesystems without reading every +byte of every file. Files are considered duplicates when their sizes are equal +and they agree on a short ladder of hashes. A file under 10 MiB is hashed in +full and compared directly. A larger file is gated first on the SHA-256 of its +first 64 KiB and of its last 64 KiB, and only when its size and both of those +match another file's is it read for a content hash to compare — the SHA-256 of +the whole file when it is under 50 MiB, or of gigabyte-spaced 1 MiB samples when +it is 50 MiB or larger. Below 50 MiB the content hash is proof of identical +content; at or above 50 MiB it is a strong candidate signal rather than proof, +because the gaps between samples are never read. The intended use is finding +duplicate downloads and duplicated directory trees on multi-terabyte ZFS servers +where reading every byte of every file is prohibitively expensive. `scan` +maintains a persistent SQLite database of file signatures that survives between +runs, so it can be run from cron and the reports can be generated at any time +from the most recent scan. This README is the complete and authoritative specification. @@ -34,22 +32,19 @@ export SFDUPES_DATABASE="$HOME/.local/share/sfdupes/db.sqlite" ./sfdupes trees > dupetrees.tsv ``` -`scan` walks one or more filesystem trees and maintains one database -record per regular file (path, size, mtime, head hash, tail hash, -content hash). The -database persists between runs; a rescan only hashes files that are new -or changed, or that may have gained a duplicate since the last scan, -and removes records for files that no longer exist. -`report` reads the database and prints the file-level duplicates -report. `trees` reads the same database and prints the duplicate-tree -report. A missing/invalid subcommand — or a `scan` invocation with no -`PATH` operand — prints a usage message and exits 2. +`scan` walks one or more filesystem trees and maintains one database record per +regular file (path, size, mtime, head hash, tail hash, content hash). The +database persists between runs; a rescan only hashes files that are new or +changed, or that may have gained a duplicate since the last scan, and removes +records for files that no longer exist. `report` reads the database and prints +the file-level duplicates report. `trees` reads the same database and prints the +duplicate-tree report. A missing/invalid subcommand — or a `scan` invocation +with no `PATH` operand — prints a usage message and exits 2. -The database defaults to `/var/lib/sfdupes/db.sqlite` and can be placed -anywhere by setting `SFDUPES_DATABASE`. The intended deployment is a -daily `sfdupes scan` cron job, with the reporting commands run -interactively whenever needed; their results are as fresh as the last -completed scan. +The database defaults to `/var/lib/sfdupes/db.sqlite` and can be placed anywhere +by setting `SFDUPES_DATABASE`. The intended deployment is a daily `sfdupes scan` +cron job, with the reporting commands run interactively whenever needed; their +results are as fresh as the last completed scan. ### Install @@ -59,10 +54,9 @@ With Go installed, this builds and installs the current `main` branch: go install sneak.berlin/go/sfdupes@main ``` -The binary goes to `$(go env GOPATH)/bin`, or to `$GOBIN` when that is -set. A binary installed this way reports its version as `dev`; one -built from a clone or into the Docker image carries the git tag or -commit it was built from. +The binary goes to `$(go env GOPATH)/bin`, or to `$GOBIN` when that is set. A +binary installed this way reports its version as `dev`; one built from a clone +or into the Docker image carries the git tag or commit it was built from. From a clone, `make build` writes the binary to `./sfdupes`: @@ -74,13 +68,12 @@ make build Copy the binary to `/usr/local/bin` for the cron job below. -`make docker` builds the Docker image, tagged `sfdupes`, after running -the tests and the linter (see "Build"). The image runs `sfdupes` as -root with the database at its default path, so a bind mount of -`/var/lib/sfdupes` keeps the database between runs. Mount the scanned -tree at the same path inside the container as on the host; read-only -is enough. The database records paths as the container sees them, so -the reports then name the host's paths. +`make docker` builds the Docker image, tagged `sfdupes`, after running the tests +and the linter (see "Build"). The image runs `sfdupes` as root with the database +at its default path, so a bind mount of `/var/lib/sfdupes` keeps the database +between runs. Mount the scanned tree at the same path inside the container as on +the host; read-only is enough. The database records paths as the container sees +them, so the reports then name the host's paths. ```sh make docker @@ -91,52 +84,49 @@ docker run --rm -v /var/lib/sfdupes:/var/lib/sfdupes sfdupes report > dupes.tsv ### Daily scan from cron -Run `scan` as root, so that it can read every file: a path it cannot -read is skipped with a warning and loses its database record (see -"Rules for the walk"). As a file `/etc/cron.d/sfdupes`: +Run `scan` as root, so that it can read every file: a path it cannot read is +skipped with a warning and loses its database record (see "Rules for the walk"). +As a file `/etc/cron.d/sfdupes`: ``` 30 3 * * * root /usr/local/bin/sfdupes scan /srv 2>>/var/log/sfdupes.log || tail -n 3 /var/log/sfdupes.log ``` -- The database is `/var/lib/sfdupes/db.sqlite`, created with its - directory by the first scan. To keep it elsewhere, set - `SFDUPES_DATABASE=/path/to/db.sqlite` before the command on the same - line. -- `scan` writes nothing to stdout. Its stderr, appended here to - `/var/log/sfdupes.log`, holds a plain progress line as each phase - starts and then at most every 5 seconds, a warning for each path it - skips, and the summary line (see "Progress" and "`scan` mode"). The - log grows with every scan; rotate it like any other. -- Skipped paths do not fail a scan: it still exits 0, and cron sends - nothing. A scan that fails, or is stopped by `SIGINT` or `SIGTERM`, - exits 1 with the reason among the last lines of the log; `tail` - prints them, and cron mails them to root if the host can send mail. -- A scan still running when the next one starts carries on. The new - one fails at once, and the lines cron mails include - `sfdupes: another scan is running (lock held on /var/lib/sfdupes/db.sqlite.lock)`. -- `report` and `trees` need only read access to the database (see - "Database"). Under the usual umask of `022` the first scan creates - it readable by every user, so an unprivileged user can run them - against root's database. +- The database is `/var/lib/sfdupes/db.sqlite`, created with its directory by + the first scan. To keep it elsewhere, set + `SFDUPES_DATABASE=/path/to/db.sqlite` before the command on the same line. +- `scan` writes nothing to stdout. Its stderr, appended here to + `/var/log/sfdupes.log`, holds a plain progress line as each phase starts and + then at most every 5 seconds, a warning for each path it skips, and the + summary line (see "Progress" and "`scan` mode"). The log grows with every + scan; rotate it like any other. +- Skipped paths do not fail a scan: it still exits 0, and cron sends nothing. A + scan that fails, or is stopped by `SIGINT` or `SIGTERM`, exits 1 with the + reason among the last lines of the log; `tail` prints them, and cron mails + them to root if the host can send mail. +- A scan still running when the next one starts carries on. The new one fails at + once, and the lines cron mails include + `sfdupes: another scan is running (lock held on /var/lib/sfdupes/db.sqlite.lock)`. +- `report` and `trees` need only read access to the database (see "Database"). + Under the usual umask of `022` the first scan creates it readable by every + user, so an unprivileged user can run them against root's database. ### Reading the reports -Each row of `report` names two copies of one file, and each row of -`trees` two copies of one directory tree (see "Report output format" -and "Trees output format"). In a group of copies, the path that sorts -first byte by byte is `first` and every other path is a `dupe` of it. -`first` says nothing about which copy is the original or the oldest; -which copy to keep is your choice. +Each row of `report` names two copies of one file, and each row of `trees` two +copies of one directory tree (see "Report output format" and "Trees output +format"). In a group of copies, the path that sorts first byte by byte is +`first` and every other path is a `dupe` of it. `first` says nothing about which +copy is the original or the oldest; which copy to keep is your choice. A row is a candidate, not proof: -- The reports read only the database, so they show the files as of - the last scan; a file may have changed or gone since. -- A file of 50 MiB or more is compared only on samples of its content - (see "Duplicate detection"). -- Paths that are hard links to one file are listed as duplicates, but - they share their data, so removing one frees nothing. +- The reports read only the database, so they show the files as of the last + scan; a file may have changed or gone since. +- A file of 50 MiB or more is compared only on samples of its content (see + "Duplicate detection"). +- Paths that are hard links to one file are listed as duplicates, but they share + their data, so removing one frees nothing. Compare a pair byte for byte before removing either copy. For the row `/srv/a/big.iso`, `/srv/b/big-copy.iso`, `4294967296`: @@ -146,120 +136,110 @@ cmp /srv/a/big.iso /srv/b/big-copy.iso && echo identical [ /srv/a/big.iso -ef /srv/b/big-copy.iso ] && echo "hard links" ``` -`cmp` prints nothing and exits 0 only when every byte matches, and -otherwise reports where the files differ. The second line prints -`hard links` when the two paths are the same file, so removing either -frees nothing. +`cmp` prints nothing and exits 0 only when every byte matches, and otherwise +reports where the files differ. The second line prints `hard links` when the two +paths are the same file, so removing either frees nothing. -A path holding a backslash, tab, newline or carriage return is escaped -in the reports (see "Report output format"). Undo the escapes before -using it. `printf '%b'` does exactly that, because every backslash in -an escaped path starts one of the four escapes. Command substitution -drops trailing newlines, so print an `x` after the path and remove it -afterwards, or a path that ends in a newline names a different file: +A path holding a backslash, tab, newline or carriage return is escaped in the +reports (see "Report output format"). Undo the escapes before using it. +`printf '%b'` does exactly that, because every backslash in an escaped path +starts one of the four escapes. Command substitution drops trailing newlines, so +print an `x` after the path and remove it afterwards, or a path that ends in a +newline names a different file: ```sh p="$(printf '%bx' '/srv/a/tab\tname.txt')"; p="${p%x}" cmp "$p" /srv/b/tab-copy.txt ``` -Check a `trees` row with `diff -r`, which compares the two trees file -by file and also names anything present in only one of them, such as -an empty directory or a symlink, which `trees` does not see. +Check a `trees` row with `diff -r`, which compares the two trees file by file +and also names anything present in only one of them, such as an empty directory +or a symlink, which `trees` does not see. ## Rationale -Duplicate finders that hash entire files do not scale to the target -environment: ~10 million files and ~150 TB on possibly slow or busy -disks (a ZFS pool under resilver). sfdupes spends disk I/O only on files -whose size at least one other file shares, since a size-unique file -cannot be a duplicate. Of those, a file under 10 MiB is read in full; a -larger one has its cheap end windows read first, and is read for a -content hash only when its size and both end windows match another -file's — the whole file below 50 MiB, but only gigabyte-spaced samples -at or above 50 MiB, so the largest files are never read in full. This -keeps a full-filesystem sweep tractable, and the signatures are kept in -a persistent database, so the expensive filesystem pass is incremental: -a rescan re-hashes only files whose recorded mtime or size changed, plus -— for its content hash — a file of 10 MiB or more whose size and end -windows have come to match another file's. All analysis happens offline -from the database alone. The end goal is -not individual files but whole duplicated trees — duplicate -extractions, duplicate downloads, copied project trees — which an -operator can consider removing as a unit. +Duplicate finders that hash entire files do not scale to the target environment: +~10 million files and ~150 TB on possibly slow or busy disks (a ZFS pool under +resilver). sfdupes spends disk I/O only on files whose size at least one other +file shares, since a size-unique file cannot be a duplicate. Of those, a file +under 10 MiB is read in full; a larger one has its cheap end windows read first, +and is read for a content hash only when its size and both end windows match +another file's — the whole file below 50 MiB, but only gigabyte-spaced samples +at or above 50 MiB, so the largest files are never read in full. This keeps a +full-filesystem sweep tractable, and the signatures are kept in a persistent +database, so the expensive filesystem pass is incremental: a rescan re-hashes +only files whose recorded mtime or size changed, plus — for its content hash — a +file of 10 MiB or more whose size and end windows have come to match another +file's. All analysis happens offline from the database alone. The end goal is +not individual files but whole duplicated trees — duplicate extractions, +duplicate downloads, copied project trees — which an operator can consider +removing as a unit. ## Design Goals, in order: -1. **Find whole duplicate trees, not just files.** The end goal is to - identify places where the exact same set of files and directories - exists at two or more paths (duplicate extractions, duplicate - downloads, copied project trees), so the operator can consider - removing an entire subtree at once. File-level duplicate detection is - the foundation; tree-level detection is built on top of it. -2. **Spend I/O in proportion to duplicate likelihood.** Only files - whose size at least one other file shares are read at all — a - size-unique file cannot be a duplicate. Those are compared by the - ladder in "Duplicate detection" below: a file under 10 MiB is hashed - in full, while a larger file is gated on cheap 64 KiB end windows - first, and gets a content hash only when its size and both end - windows match another file's. That hash reads the whole file below - 50 MiB but only gigabyte-spaced 1 MiB samples at or above it, so the - very largest files are still never read in full. Scale target: tens of - millions of files, ~150 TB filesystem, possibly slow or busy disks - (ZFS pool under resilver). Holding one small record (path, size, - mtime) per file in memory during a scan is acceptable; holding - every file's hashes is not (they stay in the database). The - reporting commands do not hold every file's hashes either: `report` - lets SQLite group and order the records and writes each row as it - reads it, so its memory does not grow with the database, and - `trees` reads the records in path order and keeps each directory's - path, digest and totals, plus the hashes of only the files in the - directories holding the record being read, so its memory grows with - the number of directories and with the size of the largest - directory. -3. **Scan incrementally, analyze offline.** The expensive filesystem - scan maintains a persistent database; an unchanged file is never - read again on a rescan, except to compute its content hash once a - file of 10 MiB or more comes to match another on size and both end - windows. All analysis (`report`, `trees`) works from - the database alone and must never touch the scanned filesystem - again. `scan` is designed to be cronned; the reports run at any +1. **Find whole duplicate trees, not just files.** The end goal is to identify + places where the exact same set of files and directories exists at two or + more paths (duplicate extractions, duplicate downloads, copied project + trees), so the operator can consider removing an entire subtree at once. + File-level duplicate detection is the foundation; tree-level detection is + built on top of it. +2. **Spend I/O in proportion to duplicate likelihood.** Only files whose size + at least one other file shares are read at all — a size-unique file cannot + be a duplicate. Those are compared by the ladder in "Duplicate detection" + below: a file under 10 MiB is hashed in full, while a larger file is gated + on cheap 64 KiB end windows first, and gets a content hash only when its + size and both end windows match another file's. That hash reads the whole + file below 50 MiB but only gigabyte-spaced 1 MiB samples at or above it, so + the very largest files are still never read in full. Scale target: tens of + millions of files, ~150 TB filesystem, possibly slow or busy disks (ZFS pool + under resilver). Holding one small record (path, size, mtime) per file in + memory during a scan is acceptable; holding every file's hashes is not (they + stay in the database). The reporting commands do not hold every file's + hashes either: `report` lets SQLite group and order the records and writes + each row as it reads it, so its memory does not grow with the database, and + `trees` reads the records in path order and keeps each directory's path, + digest and totals, plus the hashes of only the files in the directories + holding the record being read, so its memory grows with the number of + directories and with the size of the largest directory. +3. **Scan incrementally, analyze offline.** The expensive filesystem scan + maintains a persistent database; an unchanged file is never read again on a + rescan, except to compute its content hash once a file of 10 MiB or more + comes to match another on size and both end windows. All analysis (`report`, + `trees`) works from the database alone and must never touch the scanned + filesystem again. `scan` is designed to be cronned; the reports run at any time against the last completed scan. -4. **Clean stream separation.** Everything on stdout is machine-readable - data. All progress, warnings, summaries, and help and usage text go - to stderr. Never mix them. +4. **Clean stream separation.** Everything on stdout is machine-readable data. + All progress, warnings, summaries, and help and usage text go to stderr. + Never mix them. ### Constraints -- Language: Go (module `sneak.berlin/go/sfdupes`). Binary name: - `sfdupes`. -- Dependencies: standard library, `github.com/spf13/cobra` for the - CLI, **one progress-bar library** - (`github.com/schollz/progressbar/v3`), `golang.org/x/term` to tell - whether stderr is a terminal, **one SQLite driver** - (`modernc.org/sqlite`, pure Go, so builds keep cgo disabled), and - `golang.org/x/sys` for `flock(2)` (the scan lock, see "Database"). - `github.com/spf13/viper` is permitted if configuration-file support - is ever needed, but is not currently used. No other third-party - deps. -- Cross-compilation is not a concern. Builds run with cgo disabled (the - `Makefile` exports `CGO_ENABLED=0`); the code must remain pure Go. -- Analysis modes (`report`, `trees`) must be deterministic: identical - database contents, identical output, regardless of the order in - which records were inserted. +- Language: Go (module `sneak.berlin/go/sfdupes`). Binary name: `sfdupes`. +- Dependencies: standard library, `github.com/spf13/cobra` for the CLI, **one + progress-bar library** (`github.com/schollz/progressbar/v3`), + `golang.org/x/term` to tell whether stderr is a terminal, **one SQLite + driver** (`modernc.org/sqlite`, pure Go, so builds keep cgo disabled), and + `golang.org/x/sys` for `flock(2)` (the scan lock, see "Database"). + `github.com/spf13/viper` is permitted if configuration-file support is ever + needed, but is not currently used. No other third-party deps. +- Cross-compilation is not a concern. Builds run with cgo disabled (the + `Makefile` exports `CGO_ENABLED=0`); the code must remain pure Go. +- Analysis modes (`report`, `trees`) must be deterministic: identical database + contents, identical output, regardless of the order in which records were + inserted. ### Subcommands Three subcommands, all implemented: -1. `scan` — walk the filesystem and synchronize the database: one - signature record per regular file. +1. `scan` — walk the filesystem and synchronize the database: one signature + record per regular file. 2. `report` — file-level duplicate report from the database. -3. `trees` — tree-level duplicate report: reconstruct the directory - hierarchy from the database records, compute a Merkle-style digest - per directory, and report maximal groups of identical trees. +3. `trees` — tree-level duplicate report: reconstruct the directory hierarchy + from the database records, compute a Merkle-style digest per directory, and + report maximal groups of identical trees. ``` sfdupes scan [--workers N] [-x] PATH... @@ -269,72 +249,64 @@ sfdupes --version sfdupes [command] --help ``` -`--workers N` sets the size of each `scan` worker pool (default: the -number of CPUs), and `-x` (`--one-file-system`) keeps the walk of each -operand on that operand's filesystem; both are described under -"`scan` mode". `sfdupes --version` (or `-v`) prints one line, -`sfdupes VERSION`, to stdout and exits 0, writing nothing to stderr. -`-h` or `--help`, alone or after a subcommand, prints the help text to -stderr and exits 0, writing nothing to stdout. +`--workers N` sets the size of each `scan` worker pool (default: the number of +CPUs), and `-x` (`--one-file-system`) keeps the walk of each operand on that +operand's filesystem; both are described under "`scan` mode". +`sfdupes --version` (or `-v`) prints one line, `sfdupes VERSION`, to stdout and +exits 0, writing nothing to stderr. `-h` or `--help`, alone or after a +subcommand, prints the help text to stderr and exits 0, writing nothing to +stdout. ### Database All three subcommands operate on a single SQLite database file: -- Location: the value of the `SFDUPES_DATABASE` environment variable - when set and non-empty, otherwise `/var/lib/sfdupes/db.sqlite`. - There is no command-line flag. -- `scan` creates the database (and its parent directory) on first - use. `report` and `trees` require an existing database; a missing - database file is a fatal error (exit 1) telling the user to run - `scan` first. -- Only one `scan` runs against a database at a time. For its whole - run, `scan` holds an exclusive `flock(2)` lock on a lock file - beside the database, named by appending `.lock` to the database - path (`/var/lib/sfdupes/db.sqlite.lock` by default), taken before - it walks the filesystem or opens the database. A second `scan` - against the same database does not wait: it fails at once with a - one-line error naming the lock file and exits 1, without walking - anything or opening the database, and the running scan carries on. - The lock file is created on first use, open to its owner only, and - left in place: a leftover file blocks nothing, because the lock - ends with the process holding it however it ends, a fatal error or - an interrupt included, and deleting the file while a scan runs - would let a second scan start. `report` and `trees` never take the - lock, so they run during a scan. -- While `scan` runs, the database is in WAL journal mode with a busy - timeout, so running a report while a cron `scan` is in progress is - safe. The filesystem is authoritative; the database is an - eventually-consistent reflection of it. Hashed records are - committed in batched transactions while the scan is still running - (keeping the WAL small and letting concurrent reports observe - progress), so a report may see a scan's changes partially applied, - and a scan that dies partway leaves a valid database holding - every batch committed so far (an interrupted scan also commits the - batch in progress, see "Error handling and exit codes"); the next - scan skips those records and converges toward the filesystem. -- `scan` switches the database back to rollback-journal mode when it - closes it, so between scans the database file alone holds the whole - database. Each switch needs the database to itself: a `scan` that - starts while a report is still reading waits for it up to the - 10-second busy timeout, then fails; a `scan` that ends while a - report has the database open warns and leaves the database in WAL - mode until the next scan. `report` writes each row as it reads it, - so it is still reading while its output is paused (a pager, a - stalled pipe), and a `scan` started then fails after the busy - timeout. -- `report` and `trees` open the database read-only and need only read - access to the database file, and no write access to its directory. - While the database is in WAL mode they also read the `-wal` and - `-shm` files beside it, which SQLite creates with the database - file's permissions. -- Schema (`PRAGMA user_version` is the schema version, currently 1; a - database with any other version is a fatal error. `scan` creates - the schema and sets the version in one transaction, so a first scan - stopped while doing so leaves an empty database the next scan sets - up. A database at version 0 that already has a `files` table was - therefore not made by sfdupes; every subcommand refuses it with an - error telling the user to remove the file and rescan): +- Location: the value of the `SFDUPES_DATABASE` environment variable when set + and non-empty, otherwise `/var/lib/sfdupes/db.sqlite`. There is no + command-line flag. +- `scan` creates the database (and its parent directory) on first use. `report` + and `trees` require an existing database; a missing database file is a fatal + error (exit 1) telling the user to run `scan` first. +- Only one `scan` runs against a database at a time. For its whole run, `scan` + holds an exclusive `flock(2)` lock on a lock file beside the database, named + by appending `.lock` to the database path (`/var/lib/sfdupes/db.sqlite.lock` + by default), taken before it walks the filesystem or opens the database. A + second `scan` against the same database does not wait: it fails at once with a + one-line error naming the lock file and exits 1, without walking anything or + opening the database, and the running scan carries on. The lock file is + created on first use, open to its owner only, and left in place: a leftover + file blocks nothing, because the lock ends with the process holding it however + it ends, a fatal error or an interrupt included, and deleting the file while a + scan runs would let a second scan start. `report` and `trees` never take the + lock, so they run during a scan. +- While `scan` runs, the database is in WAL journal mode with a busy timeout, so + running a report while a cron `scan` is in progress is safe. The filesystem is + authoritative; the database is an eventually-consistent reflection of it. + Hashed records are committed in batched transactions while the scan is still + running (keeping the WAL small and letting concurrent reports observe + progress), so a report may see a scan's changes partially applied, and a scan + that dies partway leaves a valid database holding every batch committed so far + (an interrupted scan also commits the batch in progress, see "Error handling + and exit codes"); the next scan skips those records and converges toward the + filesystem. +- `scan` switches the database back to rollback-journal mode when it closes it, + so between scans the database file alone holds the whole database. Each switch + needs the database to itself: a `scan` that starts while a report is still + reading waits for it up to the 10-second busy timeout, then fails; a `scan` + that ends while a report has the database open warns and leaves the database + in WAL mode until the next scan. `report` writes each row as it reads it, so + it is still reading while its output is paused (a pager, a stalled pipe), and + a `scan` started then fails after the busy timeout. +- `report` and `trees` open the database read-only and need only read access to + the database file, and no write access to its directory. While the database is + in WAL mode they also read the `-wal` and `-shm` files beside it, which SQLite + creates with the database file's permissions. +- Schema (`PRAGMA user_version` is the schema version, currently 1; a database + with any other version is a fatal error. `scan` creates the schema and sets + the version in one transaction, so a first scan stopped while doing so leaves + an empty database the next scan sets up. A database at version 0 that already + has a `files` table was therefore not made by sfdupes; every subcommand + refuses it with an error telling the user to remove the file and rescan): ```sql CREATE TABLE files ( @@ -348,274 +320,242 @@ All three subcommands operate on a single SQLite database file: CREATE INDEX files_signature ON files (size, head, tail, content); ``` - Paths are stored as BLOBs because Unix paths are raw bytes, not - guaranteed UTF-8. `mtime` is used only for change detection; it is - not part of the duplicate key. For a file under 10 MiB `head`, `tail`, - and `content` all hold the whole-file hash (that range is hashed in - full, with no end windows); for a larger file `head` and `tail` hold - the first- and last-64 KiB hashes and `content` the whole-file or - sampled hash. All three are empty strings when the file has never - been hashed because its size was unique as of the last scan that - covered it. For a file of 10 MiB or more, `content` stays empty - until the content phase of a scan (see "`scan` mode" below) has - read the file. A record with an empty `content` is never part of a - duplicate group, though it still defines the file for tree - reconstruction. The `files_signature` index lets SQLite group the - records by signature for `report` without sorting the whole table. + Paths are stored as BLOBs because Unix paths are raw bytes, not guaranteed + UTF-8. `mtime` is used only for change detection; it is not part of the + duplicate key. For a file under 10 MiB `head`, `tail`, and `content` all + hold the whole-file hash (that range is hashed in full, with no end + windows); for a larger file `head` and `tail` hold the first- and last-64 + KiB hashes and `content` the whole-file or sampled hash. All three are empty + strings when the file has never been hashed because its size was unique as + of the last scan that covered it. For a file of 10 MiB or more, `content` + stays empty until the content phase of a scan (see "`scan` mode" below) has + read the file. A record with an empty `content` is never part of a duplicate + group, though it still defines the file for tree reconstruction. The + `files_signature` index lets SQLite group the records by signature for + `report` without sorting the whole table. ### Duplicate detection -Two files are duplicates only when they agree on every rung of this -ladder; a mismatch at any rung means they are not duplicates. `scan` -stores each file's hashes, and `report` and `trees` group files by the -whole signature — size, `head`, `tail`, and `content` — so the grouping -is exactly this ladder applied across everything scanned into the -database, even across separate scans. +Two files are duplicates only when they agree on every rung of this ladder; a +mismatch at any rung means they are not duplicates. `scan` stores each file's +hashes, and `report` and `trees` group files by the whole signature — size, +`head`, `tail`, and `content` — so the grouping is exactly this ladder applied +across everything scanned into the database, even across separate scans. -1. **Size.** Files of different sizes are never compared. Only files - whose size at least one other file shares are hashed at all. -2. **Under 10 MiB: whole file.** A file smaller than 10 MiB is hashed - in full and compared directly, with no separate end-window step — - small files are cheap to read to the last byte, and doing so makes - the comparison exact. `head`, `tail`, and `content` all hold this - whole-file SHA-256, so such a file's signature is decided entirely - by its size and its content. -3. **10 MiB and above: head and tail.** For a larger file, the SHA-256 - of the first 64 KiB (`head`) and of the last 64 KiB (`tail`) are a - cheap gate that eliminates most same-size pairs before any bulk - reading: the content hash of the next two rungs is computed only for - a file whose size, `head`, and `tail` match another file's, whether - that file is scanned in the same run or stored by an earlier scan. - A stored file that first gains such a match in a later scan gets its - content hash then; until it has one, its `content` is empty and it - is not a duplicate. At 10 MiB and above the two windows never +1. **Size.** Files of different sizes are never compared. Only files whose size + at least one other file shares are hashed at all. +2. **Under 10 MiB: whole file.** A file smaller than 10 MiB is hashed in full + and compared directly, with no separate end-window step — small files are + cheap to read to the last byte, and doing so makes the comparison exact. + `head`, `tail`, and `content` all hold this whole-file SHA-256, so such a + file's signature is decided entirely by its size and its content. +3. **10 MiB and above: head and tail.** For a larger file, the SHA-256 of the + first 64 KiB (`head`) and of the last 64 KiB (`tail`) are a cheap gate that + eliminates most same-size pairs before any bulk reading: the content hash of + the next two rungs is computed only for a file whose size, `head`, and + `tail` match another file's, whether that file is scanned in the same run or + stored by an earlier scan. A stored file that first gains such a match in a + later scan gets its content hash then; until it has one, its `content` is + empty and it is not a duplicate. At 10 MiB and above the two windows never overlap. -4. **10 MiB and above, content below 50 MiB.** The SHA-256 of the - entire file. Agreement here is proof of identical content (barring a - SHA-256 collision). -5. **10 MiB and above, content 50 MiB and above.** A sampled SHA-256: - the 1 MiB window at each gigabyte-aligned offset (0, 1 GiB, 2 GiB, … - while inside the file, the final window truncated at end of file) is - fed, in order, into one hash. This is **deliberately probabilistic** - — the gaps between samples are never read, so two large files that - agree on every sample are reported as duplicates without being read - in full. It is the price of never reading a 150 GB file end to end. - Because size is already part of the signature, only equal-size files - reach this rung, so their sample boundaries always align. +4. **10 MiB and above, content below 50 MiB.** The SHA-256 of the entire file. + Agreement here is proof of identical content (barring a SHA-256 collision). +5. **10 MiB and above, content 50 MiB and above.** A sampled SHA-256: the 1 MiB + window at each gigabyte-aligned offset (0, 1 GiB, 2 GiB, … while inside the + file, the final window truncated at end of file) is fed, in order, into one + hash. This is **deliberately probabilistic** — the gaps between samples are + never read, so two large files that agree on every sample are reported as + duplicates without being read in full. It is the price of never reading a + 150 GB file end to end. Because size is already part of the signature, only + equal-size files reach this rung, so their sample boundaries always align. -`head`, `tail`, and `content` are one column each. A file below 10 MiB -and one at or above it never share a size, and neither do a file below -50 MiB and one at or above it, so a stored value is never ambiguous -between the whole-file, end-window, and sampled forms. +`head`, `tail`, and `content` are one column each. A file below 10 MiB and one +at or above it never share a size, and neither do a file below 50 MiB and one at +or above it, so a stored value is never ambiguous between the whole-file, +end-window, and sampled forms. ### `scan` mode -`scan` requires one or more `PATH` operands naming the trees to scan. -There is no default path; invoking `scan` with no operand is a usage -error (usage message on stderr, exit 2). An operand may be a directory -or a regular file; an operand that does not exist is a fatal error -(exit 1). Because database records persist between runs and are keyed -by absolute path, each operand is resolved to an absolute, lexically -cleaned path (symlinks are not resolved) before walking, so results do -not depend on the working directory. All operands belong to a single -scan and are enumerated concurrently: every operand seeds the shared -walk worker pool. Overlapping operands are harmless — an operand that -duplicates another or lies under another is dropped before walking, -so every file is reached exactly once and produces one database -record. +`scan` requires one or more `PATH` operands naming the trees to scan. There is +no default path; invoking `scan` with no operand is a usage error (usage message +on stderr, exit 2). An operand may be a directory or a regular file; an operand +that does not exist is a fatal error (exit 1). Because database records persist +between runs and are keyed by absolute path, each operand is resolved to an +absolute, lexically cleaned path (symlinks are not resolved) before walking, so +results do not depend on the working directory. All operands belong to a single +scan and are enumerated concurrently: every operand seeds the shared walk worker +pool. Overlapping operands are harmless — an operand that duplicates another or +lies under another is dropped before walking, so every file is reached exactly +once and produces one database record. -An operand that is a symlink (never followed, not even as an operand), -socket, FIFO, or device node, or a directory named `.zfs`, is not -scanned. `scan` prints a one-line warning naming the path and what it -is, counts it as skipped, and drops it from the scanned operands before -reading the database. Another operand beneath it is still scanned. The -records stored beneath it are not deleted: they are treated like any -other record outside the scanned operands, including the content-phase -exception below. If it lies under another operand, they are under that -operand instead, and are deleted like any other record there that this -scan did not verify. This is not an error: a scan whose every operand -is dropped walks nothing and exits 0. +An operand that is a symlink (never followed, not even as an operand), socket, +FIFO, or device node, or a directory named `.zfs`, is not scanned. `scan` prints +a one-line warning naming the path and what it is, counts it as skipped, and +drops it from the scanned operands before reading the database. Another operand +beneath it is still scanned. The records stored beneath it are not deleted: they +are treated like any other record outside the scanned operands, including the +content-phase exception below. If it lies under another operand, they are under +that operand instead, and are deleted like any other record there that this scan +did not verify. This is not an error: a scan whose every operand is dropped +walks nothing and exits 0. -`scan` synchronizes the database with the filesystem state under the -scanned operands: +`scan` synchronizes the database with the filesystem state under the scanned +operands: -- Only a file whose size at least one other file shares is ever - read: a size-unique file cannot be a duplicate, so it is recorded - without hashes (`head`, `tail`, and `content` empty). The size - census covers every file walked this scan plus every database - record outside the scanned operands, so a possible duplicate of a - separately scanned tree is still recognized. -- A file not yet in the database is inserted: hashed when its size - is shared, without hashes otherwise. -- A file already in the database is **skipped without reading its - contents** when its lstat size equals the recorded size and its - lstat mtime is not newer than the recorded mtime. This is what - makes a daily rescan cheap. Exception: an unchanged file whose - record lacks hashes is hashed — and its record updated — once its - size becomes shared, so hashing deferred by size-uniqueness - happens as soon as it could matter. Likewise, an unchanged file of - 10 MiB or more whose record has no `content` hash is read for one - by the content phase below once its size, `head`, and `tail` match - another record's. -- A file whose mtime is newer than recorded, or whose size differs, - is processed as if new: re-hashed, or recorded without hashes, - per the shared-size rule. -- A database record whose path lies under one of the scanned operands - but was not successfully processed this run is deleted. This - removes records for deleted files. It also removes records for - paths that failed to stat or hash this run: the database only ever - contains signatures verified by the most recent scan that covered - them (a subsequent successful scan re-adds such files). A failure - in the content phase below removes nothing: the record is left as - it is. -- Database records outside the scanned operands are untouched, so - disjoint trees can be scanned on different schedules into the same - database. The one exception is the content phase below: a stored - file of 10 MiB or more without a `content` hash is read for one, - wherever it lies, once its size, `head`, and `tail` match another - record's. If that file is gone or has changed since its record was - written, the record is left as it is. +- Only a file whose size at least one other file shares is ever read: a + size-unique file cannot be a duplicate, so it is recorded without hashes + (`head`, `tail`, and `content` empty). The size census covers every file + walked this scan plus every database record outside the scanned operands, so a + possible duplicate of a separately scanned tree is still recognized. +- A file not yet in the database is inserted: hashed when its size is shared, + without hashes otherwise. +- A file already in the database is **skipped without reading its contents** + when its lstat size equals the recorded size and its lstat mtime is not newer + than the recorded mtime. This is what makes a daily rescan cheap. Exception: + an unchanged file whose record lacks hashes is hashed — and its record updated + — once its size becomes shared, so hashing deferred by size-uniqueness happens + as soon as it could matter. Likewise, an unchanged file of 10 MiB or more + whose record has no `content` hash is read for one by the content phase below + once its size, `head`, and `tail` match another record's. +- A file whose mtime is newer than recorded, or whose size differs, is processed + as if new: re-hashed, or recorded without hashes, per the shared-size rule. +- A database record whose path lies under one of the scanned operands but was + not successfully processed this run is deleted. This removes records for + deleted files. It also removes records for paths that failed to stat or hash + this run: the database only ever contains signatures verified by the most + recent scan that covered them (a subsequent successful scan re-adds such + files). A failure in the content phase below removes nothing: the record is + left as it is. +- Database records outside the scanned operands are untouched, so disjoint trees + can be scanned on different schedules into the same database. The one + exception is the content phase below: a stored file of 10 MiB or more without + a `content` hash is read for one, wherever it lies, once its size, `head`, and + `tail` match another record's. If that file is gone or has changed since its + record was written, the record is left as it is. -`scan` runs **four sequential phases over the whole scan**. -Parallelism lives inside each phase; batched database writes begin -during the hash phase: +`scan` runs **four sequential phases over the whole scan**. Parallelism lives +inside each phase; batched database writes begin during the hash phase: -1. **walk + stat** — enumerate the trees under all `PATH` operands - concurrently with the walk worker pool: every operand seeds the - shared queue, and each worker reads one directory at a time, - handing discovered subdirectories back to the queue and running - `lstat` on each regular file as it is discovered (while the - directory's metadata is still hot). Sequential directory - enumeration is metadata-latency-bound and takes hours at tens of - millions of files; per-directory parallelism is what makes the - walk tractable on large or busy pools. The walk builds the size - census and resolves unchanged already-hashed files on the fly; - every other file is carried to the hash phase as a (path, size, - mtime) record. -2. **hash** — with the census complete, each carried file's size - decides its fate. Size-unique files are never read: new or - changed ones are recorded without hashes in the update phase, - unchanged unhashed ones simply keep their records. Every file - with a shared size is hashed by the worker pool as described in - "Duplicate detection" above: a file under 10 MiB in full, which - gives its `head`, `tail`, and `content` alike, and a larger file - only in its end windows, which give its `head` and `tail`; its - content hash is left to the content phase. Zero-length files have - constant hashes and are never opened. Files are hashed in **inode - order** (minimizing seeks on spinning disks), and paths that are hard - links to the same inode are **read once**, all sharing the one - result — a hard-link backup farm costs one read per inode, not - per path. The phase total counts actual reads, so progress and - ETA are meaningful. Completed records are committed in batched - transactions **while hashing runs**, so a scan interrupted after - hours keeps everything hashed so far and the next scan resumes - cheaply, skipping records already written. -3. **update** — commit the final partial batch, the hash-less - records for size-unique new and changed files, and the deletions - for records the scan did not verify (vanished files, plus paths - that failed to stat or hash). -4. **content** — find every record of 10 MiB or more without a - `content` hash whose size, `head`, and `tail` equal another - record's, anywhere in the database: records from this scan and - records stored by earlier scans, inside or outside the scanned - operands. SQLite finds them, so only the records to be read are - kept in memory, never every file's hashes. Every record sharing - their size, `head`, and `tail`, including one that already has a - `content` hash, has its file checked with `lstat` first. A file - that is gone, is no longer a regular file, or has changed (a - different size, or an mtime newer than recorded) keeps its record - as it is and does not count as a match for the others. Any other - `lstat` error is warned about and counted as skipped, with the same - result. If such a record has no `content` hash, it stays out of - duplicate groups; if it has one, it is still reported until a scan - covering its own tree updates or removes it. The files that pass - and have no `content` hash are read only if at least two of those - records pass, so a file whose only matches are stale costs no read; - a file that already has a `content` hash is never read again. - They are read by a worker pool as in the hash phase, in inode order - and once per inode, and their content hashes are committed in - batches. A failed read is warned about and counted as skipped; its - record keeps an empty `content`, so it is not a duplicate, and a - later scan tries again. +1. **walk + stat** — enumerate the trees under all `PATH` operands concurrently + with the walk worker pool: every operand seeds the shared queue, and each + worker reads one directory at a time, handing discovered subdirectories back + to the queue and running `lstat` on each regular file as it is discovered + (while the directory's metadata is still hot). Sequential directory + enumeration is metadata-latency-bound and takes hours at tens of millions of + files; per-directory parallelism is what makes the walk tractable on large + or busy pools. The walk builds the size census and resolves unchanged + already-hashed files on the fly; every other file is carried to the hash + phase as a (path, size, mtime) record. +2. **hash** — with the census complete, each carried file's size decides its + fate. Size-unique files are never read: new or changed ones are recorded + without hashes in the update phase, unchanged unhashed ones simply keep + their records. Every file with a shared size is hashed by the worker pool as + described in "Duplicate detection" above: a file under 10 MiB in full, which + gives its `head`, `tail`, and `content` alike, and a larger file only in its + end windows, which give its `head` and `tail`; its content hash is left to + the content phase. Zero-length files have constant hashes and are never + opened. Files are hashed in **inode order** (minimizing seeks on spinning + disks), and paths that are hard links to the same inode are **read once**, + all sharing the one result — a hard-link backup farm costs one read per + inode, not per path. The phase total counts actual reads, so progress and + ETA are meaningful. Completed records are committed in batched transactions + **while hashing runs**, so a scan interrupted after hours keeps everything + hashed so far and the next scan resumes cheaply, skipping records already + written. +3. **update** — commit the final partial batch, the hash-less records for + size-unique new and changed files, and the deletions for records the scan + did not verify (vanished files, plus paths that failed to stat or hash). +4. **content** — find every record of 10 MiB or more without a `content` hash + whose size, `head`, and `tail` equal another record's, anywhere in the + database: records from this scan and records stored by earlier scans, inside + or outside the scanned operands. SQLite finds them, so only the records to + be read are kept in memory, never every file's hashes. Every record sharing + their size, `head`, and `tail`, including one that already has a `content` + hash, has its file checked with `lstat` first. A file that is gone, is no + longer a regular file, or has changed (a different size, or an mtime newer + than recorded) keeps its record as it is and does not count as a match for + the others. Any other `lstat` error is warned about and counted as skipped, + with the same result. If such a record has no `content` hash, it stays out + of duplicate groups; if it has one, it is still reported until a scan + covering its own tree updates or removes it. The files that pass and have no + `content` hash are read only if at least two of those records pass, so a + file whose only matches are stale costs no read; a file that already has a + `content` hash is never read again. They are read by a worker pool as in the + hash phase, in inode order and once per inode, and their content hashes are + committed in batches. A failed read is warned about and counted as skipped; + its record keeps an empty `content`, so it is not a duplicate, and a later + scan tries again. Rules for the walk: -- Only regular files. Skip directories, symlinks (do not follow, - including symlink operands), sockets, FIFOs, and device nodes. An - operand that is a symlink, socket, FIFO, or device node is dropped - as described in "`scan` mode" above. -- Never descend into a directory named `.zfs` (ZFS snapshot pseudo-dirs; - walking them would list every file once per snapshot), not even - when it is an operand; such an operand is dropped the same way. -- Filesystem boundaries are crossed by default. With `-x` - (long form `--one-file-system`, following the GNU `du`/`rsync` - convention), never descend into a directory on a different - filesystem than its `PATH` operand; each operand is bounded by its - own filesystem. -- On any per-path error (permission denied, file vanished between - passes, unreadable): print a one-line warning to stderr, skip the - path, and continue. Per-file errors never abort the run; the final - summary reports how many were skipped. As specified above, a - skipped path that has a database record from an earlier scan loses - that record, unless it failed only in the content phase, or is an - operand dropped before the database was read that lies under no - other operand; an unreadable directory subtree likewise loses its - records (accepted: the database mirrors what the latest scan could - actually verify). +- Only regular files. Skip directories, symlinks (do not follow, including + symlink operands), sockets, FIFOs, and device nodes. An operand that is a + symlink, socket, FIFO, or device node is dropped as described in "`scan` mode" + above. +- Never descend into a directory named `.zfs` (ZFS snapshot pseudo-dirs; walking + them would list every file once per snapshot), not even when it is an operand; + such an operand is dropped the same way. +- Filesystem boundaries are crossed by default. With `-x` (long form + `--one-file-system`, following the GNU `du`/`rsync` convention), never descend + into a directory on a different filesystem than its `PATH` operand; each + operand is bounded by its own filesystem. +- On any per-path error (permission denied, file vanished between passes, + unreadable): print a one-line warning to stderr, skip the path, and continue. + Per-file errors never abort the run; the final summary reports how many were + skipped. As specified above, a skipped path that has a database record from an + earlier scan loses that record, unless it failed only in the content phase, or + is an operand dropped before the database was read that lies under no other + operand; an unreadable directory subtree likewise loses its records (accepted: + the database mirrors what the latest scan could actually verify). -Concurrency: the walk phase (which also stats files), the hash phase, -and the content phase each use a worker pool of `--workers` workers -(default `runtime.NumCPU()`); the walk parallelizes across -directories, hashing across files. `--workers` must be at least 1: a -smaller value is a usage error, reported in one line on stderr with -exit 2 before anything is scanned. All three phases are seek-bound on -spinning disks, so raising `--workers` well past the core count can -help on pools with many spindles. The main goroutine owns -partitioning, database writes, and progress rendering; progress -display must never block the workers. +Concurrency: the walk phase (which also stats files), the hash phase, and the +content phase each use a worker pool of `--workers` workers (default +`runtime.NumCPU()`); the walk parallelizes across directories, hashing across +files. `--workers` must be at least 1: a smaller value is a usage error, +reported in one line on stderr with exit 2 before anything is scanned. All three +phases are seek-bound on spinning disks, so raising `--workers` well past the +core count can help on pools with many spindles. The main goroutine owns +partitioning, database writes, and progress rendering; progress display must +never block the workers. -`scan` writes nothing to stdout. The summary line on stderr reports the -files seen this run broken down by disposition, plus skips: +`scan` writes nothing to stdout. The summary line on stderr reports the files +seen this run broken down by disposition, plus skips: ``` scan: 123400 files seen (1200 added, 34 updated, 56 removed, 122166 unchanged), 3 skipped ``` -(`removed` counts deleted database records, which are not part of the -files-seen total.) +(`removed` counts deleted database records, which are not part of the files-seen +total.) ### `report` mode -`report` reads every record from the database and takes no positional -arguments. +`report` reads every record from the database and takes no positional arguments. -**`report` must never touch the filesystem being analyzed.** It does not -stat, open, or otherwise access any path that appears in the records; its -only I/O is reading the database, writing stdout/stderr, and the -temporary file SQLite sorts in when the duplicate rows do not fit in -memory. SQLite puts that file in `$SQLITE_TMPDIR` or `$TMPDIR` when set, -otherwise in `/var/tmp` (or `/tmp`), and deletes it as soon as it has -opened it. `report` must produce identical output whether or not the -scanned filesystem is still mounted. +**`report` must never touch the filesystem being analyzed.** It does not stat, +open, or otherwise access any path that appears in the records; its only I/O is +reading the database, writing stdout/stderr, and the temporary file SQLite sorts +in when the duplicate rows do not fit in memory. SQLite puts that file in +`$SQLITE_TMPDIR` or `$TMPDIR` when set, otherwise in `/var/tmp` (or `/tmp`), and +deletes it as soon as it has opened it. `report` must produce identical output +whether or not the scanned filesystem is still mounted. Processing: -- Records without a `content` hash (see "Database" above) are - excluded: their content is unknown, so they are never reported as - duplicates. -- Group the remaining records by the key - `(size, head, tail, content)`. -- Every group with two or more paths is a duplicate group. -- Within each group, sort paths lexicographically (byte order). The - first path is the group's `first`; every other path is a `dupe`. -- Order groups by size descending (biggest reclaimable space first), - tie-broken by `first` path ascending. Output must be fully - deterministic for a given database state. +- Records without a `content` hash (see "Database" above) are excluded: their + content is unknown, so they are never reported as duplicates. +- Group the remaining records by the key `(size, head, tail, content)`. +- Every group with two or more paths is a duplicate group. +- Within each group, sort paths lexicographically (byte order). The first path + is the group's `first`; every other path is a `dupe`. +- Order groups by size descending (biggest reclaimable space first), tie-broken + by `first` path ascending. Output must be fully deterministic for a given + database state. #### Report output format -TSV on stdout: a header line, then one row per duplicate file (N-1 rows -for a group of N): +TSV on stdout: a header line, then one row per duplicate file (N-1 rows for a +group of N): ``` first dupe size @@ -623,114 +563,105 @@ first dupe size /srv/a/big.iso /srv/c/big-copy2.iso 4294967296 ``` -Paths are raw bytes and may hold any byte except NUL, so the path -columns (`first` and `dupe`) are escaped to keep every row one line of -tab-separated fields: a backslash is written as `\\`, a tab as `\t`, a -newline as `\n`, and a carriage return as `\r`. Every other byte is -written unchanged, including bytes that are not valid UTF-8. Undoing -those four escapes gives back the stored path. Grouping and ordering -use the stored path, not the escaped one. The warnings `scan` prints on -stderr are escaped the same way, so each warning is one line. +Paths are raw bytes and may hold any byte except NUL, so the path columns +(`first` and `dupe`) are escaped to keep every row one line of tab-separated +fields: a backslash is written as `\\`, a tab as `\t`, a newline as `\n`, and a +carriage return as `\r`. Every other byte is written unchanged, including bytes +that are not valid UTF-8. Undoing those four escapes gives back the stored path. +Grouping and ordering use the stored path, not the escaped one. The warnings +`scan` prints on stderr are escaped the same way, so each warning is one line. -Summary to stderr: records read, number of duplicate groups, number of -dupe files, and total reclaimable bytes (sum of `size` over all dupe -rows) in human units. +Summary to stderr: records read, number of duplicate groups, number of dupe +files, and total reclaimable bytes (sum of `size` over all dupe rows) in human +units. ### `trees` mode -`trees` reads the same database as `report` (no positional arguments) -and reports **entire duplicate directory trees**: directories under -which the exact same set of relative paths exists with the exact same -file signatures. +`trees` reads the same database as `report` (no positional arguments) and +reports **entire duplicate directory trees**: directories under which the exact +same set of relative paths exists with the exact same file signatures. -**`trees` must never touch the filesystem being analyzed** — the same -rule as `report`. The directory hierarchy is reconstructed purely from -the paths in the records, split on `/`. +**`trees` must never touch the filesystem being analyzed** — the same rule as +`report`. The directory hierarchy is reconstructed purely from the paths in the +records, split on `/`. Definitions: -- A file's **signature** is `(size, head, tail, content)` — mtime is - informational and excluded. A record without a `content` hash has - unknown content: its signature is treated as unique to that file, - so a tree containing such a file never compares equal to any other - tree. -- A directory's **digest** is a SHA-256 Merkle digest computed - bottom-up: serialize the directory's child entries — for a file - child, its name and signature; for a subdirectory child, its name - and that subdirectory's digest — sort the serialized entries - byte-lexicographically, and hash the concatenation. Names are part - of the digest: two trees whose files differ only in name are *not* - duplicates. -- Two directories are **duplicate trees** when their digests are - equal. Equal digests imply equal recursive file count and equal - total byte size. +- A file's **signature** is `(size, head, tail, content)` — mtime is + informational and excluded. A record without a `content` hash has unknown + content: its signature is treated as unique to that file, so a tree containing + such a file never compares equal to any other tree. +- A directory's **digest** is a SHA-256 Merkle digest computed bottom-up: + serialize the directory's child entries — for a file child, its name and + signature; for a subdirectory child, its name and that subdirectory's digest — + sort the serialized entries byte-lexicographically, and hash the + concatenation. Names are part of the digest: two trees whose files differ only + in name are _not_ duplicates. +- Two directories are **duplicate trees** when their digests are equal. Equal + digests imply equal recursive file count and equal total byte size. -Known limitation (accepted): hard-linked paths are reported as -duplicates by `report` and count toward duplicate trees — their -content is genuinely identical — even though they share storage, so -removing one reclaims no space. Inode identity is used during the -scan to avoid redundant reads but is not persisted in the database. +Known limitation (accepted): hard-linked paths are reported as duplicates by +`report` and count toward duplicate trees — their content is genuinely identical +— even though they share storage, so removing one reclaims no space. Inode +identity is used during the scan to avoid redundant reads but is not persisted +in the database. -Known limitation (accepted): only regular files that appear in the -database define a tree. Empty directories are invisible, and a file -skipped during the scan (e.g. permission error) in one copy but not the -other will make otherwise-identical trees compare as different. +Known limitation (accepted): only regular files that appear in the database +define a tree. Empty directories are invisible, and a file skipped during the +scan (e.g. permission error) in one copy but not the other will make +otherwise-identical trees compare as different. Processing: -- Build the hierarchy, compute every directory's digest, and group - directories by digest. Every group with two or more directories is a - duplicate-tree group. -- **Report only maximal trees.** A group is suppressed when its - members' parents are pairwise distinct directories that all share a - single digest — such a group is wholly implied by its parents' (or a - further ancestor's) group. Groups containing sibling directories, or - members whose parents differ, are always reported. -- Within each group, sort paths lexicographically (byte order); the - first path is `first`, every other path is a `dupe`. -- Order groups by total tree size descending, tie-broken by `first` - path ascending. Output must be fully deterministic for a given - input. +- Build the hierarchy, compute every directory's digest, and group directories + by digest. Every group with two or more directories is a duplicate-tree group. +- **Report only maximal trees.** A group is suppressed when its members' parents + are pairwise distinct directories that all share a single digest — such a + group is wholly implied by its parents' (or a further ancestor's) group. + Groups containing sibling directories, or members whose parents differ, are + always reported. +- Within each group, sort paths lexicographically (byte order); the first path + is `first`, every other path is a `dupe`. +- Order groups by total tree size descending, tie-broken by `first` path + ascending. Output must be fully deterministic for a given input. #### Trees output format -TSV on stdout: a header line, then one row per duplicate tree (N-1 rows -for a group of N). `files` is the recursive regular-file count of one -copy of the tree; `size` is the recursive total byte size of one copy: +TSV on stdout: a header line, then one row per duplicate tree (N-1 rows for a +group of N). `files` is the recursive regular-file count of one copy of the +tree; `size` is the recursive total byte size of one copy: ``` first dupe files size /srv/a/project /srv/backup/project 3417 104857600 ``` -The `first` and `dupe` paths are escaped as described under "Report -output format". The root directory's path is `/`. +The `first` and `dupe` paths are escaped as described under "Report output +format". The root directory's path is `/`. -Summary to stderr: records read, number of duplicate-tree groups, -number of dupe trees, and total reclaimable bytes (sum of `size` over -all dupe rows) in human units. +Summary to stderr: records read, number of duplicate-tree groups, number of dupe +trees, and total reclaimable bytes (sum of `size` over all dupe rows) in human +units. ### Progress -Use the progress-bar library for all scan progress; rendering in the -style of `pv` is the model. All progress goes to stderr. +Use the progress-bar library for all scan progress; rendering in the style of +`pv` is the model. All progress goes to stderr. -Each phase gets its own display, rendered the moment the phase -starts — a scan must never look hung. Loading the existing-record -index (`load`) and the walk have no known totals while running: show -a live count, rate, and elapsed time (spinner-style, no percentage or -ETA). The content phase's display (`content`) starts the same way, -counting the records checked while SQLite finds the files to read and -`lstat` checks them, then shows a bar once reading starts. The hash -and update phases, and the content phase's reads, have exact totals — -only files that actually need hashing appear in the hash and content -totals, so their ETAs are meaningful. Required elements for the bars -with known totals: +Each phase gets its own display, rendered the moment the phase starts — a scan +must never look hung. Loading the existing-record index (`load`) and the walk +have no known totals while running: show a live count, rate, and elapsed time +(spinner-style, no percentage or ETA). The content phase's display (`content`) +starts the same way, counting the records checked while SQLite finds the files +to read and `lstat` checks them, then shows a bar once reading starts. The hash +and update phases, and the content phase's reads, have exact totals — only files +that actually need hashing appear in the hash and content totals, so their ETAs +are meaningful. Required elements for the bars with known totals: -- elapsed time -- estimated time remaining -- a `[m/n] x%` display (items processed / total items, percent) -- current rate (items/s) +- elapsed time +- estimated time remaining +- a `[m/n] x%` display (items processed / total items, percent) +- current rate (items/s) Example shape (exact layout is flexible, content is not): @@ -740,197 +671,187 @@ hash: [12345/98765] 12% |████ | 92 files/s elapsed 2:32 eta 17:54 Additional requirements: -- When stderr is not a terminal (a pipe, a file, `/dev/null`), do not - emit ANSI redraws: print a plain one-line progress update the moment - each phase starts, then no more often than every 5 seconds. -- Progress updates are driven from the main goroutine and must be - non-blocking with respect to the worker pool. On a terminal the - spinner-style displays also redraw on their own several times a - second, so their count and elapsed time stay current while a phase - waits for its next item. -- A warning printed during a phase always lands on a line of its own, - never inside the progress display. -- A bar whose phase stops short of its total, as an interrupted one - does, is left as last drawn rather than filled up. -- `report` and `trees` modes need no progress display, only their - stderr summaries. +- When stderr is not a terminal (a pipe, a file, `/dev/null`), do not emit ANSI + redraws: print a plain one-line progress update the moment each phase starts, + then no more often than every 5 seconds. +- Progress updates are driven from the main goroutine and must be non-blocking + with respect to the worker pool. On a terminal the spinner-style displays also + redraw on their own several times a second, so their count and elapsed time + stay current while a phase waits for its next item. +- A warning printed during a phase always lands on a line of its own, never + inside the progress display. +- A bar whose phase stops short of its total, as an interrupted one does, is + left as last drawn rather than filled up. +- `report` and `trees` modes need no progress display, only their stderr + summaries. ### Error handling and exit codes -- `0`: success, even if individual files were skipped with warnings. -- `1`: fatal error (e.g., a `PATH` operand does not exist, another - `scan` is already running against the same database, the database - cannot be created/opened/read/written, a missing database for - `report`/`trees`, stdout write failure), or a `scan` stopped by - `SIGINT` or `SIGTERM` (see below). -- `2`: usage error (including `scan` with no `PATH` operand, `scan` - with `--workers` below 1, and `report`/`trees` with any positional - argument). +- `0`: success, even if individual files were skipped with warnings. +- `1`: fatal error (e.g., a `PATH` operand does not exist, another `scan` is + already running against the same database, the database cannot be + created/opened/read/written, a missing database for `report`/`trees`, stdout + write failure), or a `scan` stopped by `SIGINT` or `SIGTERM` (see below). +- `2`: usage error (including `scan` with no `PATH` operand, `scan` with + `--workers` below 1, and `report`/`trees` with any positional argument). -A stdout write failure, such as a full disk, is reported in one line on -stderr and exits 1. Two cases never reach sfdupes as a failed write: +A stdout write failure, such as a full disk, is reported in one line on stderr +and exits 1. Two cases never reach sfdupes as a failed write: -- When the reader of a stdout pipe exits early, as in - `sfdupes report | head`, the next write ends sfdupes with `SIGPIPE`, - quietly and without a summary, the way `cat` or `sort` end. The - shell reports the signal (status 141 in most shells), not exit 1. -- When stdout is closed outright (`sfdupes report >&-`), the Go - runtime opens `/dev/null` in its place before sfdupes starts, so - the output is discarded and the run succeeds, as with - `> /dev/null`. +- When the reader of a stdout pipe exits early, as in `sfdupes report | head`, + the next write ends sfdupes with `SIGPIPE`, quietly and without a summary, the + way `cat` or `sort` end. The shell reports the signal (status 141 in most + shells), not exit 1. +- When stdout is closed outright (`sfdupes report >&-`), the Go runtime opens + `/dev/null` in its place before sfdupes starts, so the output is discarded and + the run succeeds, as with `> /dev/null`. -`scan` stops cleanly on `SIGINT` (Ctrl-C) or `SIGTERM`. Its workers -stop taking work, each finishing at most the directory listing or file -it is reading; the progress display is finished; and the records it has -hashed but not yet committed are committed, so the next scan does not -hash them again. Apart from that commit it starts no further writes or -deletions: records are deleted only after a complete walk, so those -under paths an interrupted walk never reached are kept. The database is -closed and the lock released as on any other exit, the line -`scan: interrupted after N files` goes to stderr, N being the number of -files the walk reached, and the exit code is 1. The next scan skips the -records already written and converges as usual. +`scan` stops cleanly on `SIGINT` (Ctrl-C) or `SIGTERM`. Its workers stop taking +work, each finishing at most the directory listing or file it is reading; the +progress display is finished; and the records it has hashed but not yet +committed are committed, so the next scan does not hash them again. Apart from +that commit it starts no further writes or deletions: records are deleted only +after a complete walk, so those under paths an interrupted walk never reached +are kept. The database is closed and the lock released as on any other exit, the +line `scan: interrupted after N files` goes to stderr, N being the number of +files the walk reached, and the exit code is 1. The next scan skips the records +already written and converges as usual. -After the first signal `scan` stops catching them, so a second one ends -it at once, as an uncaught signal does: the records not yet committed -are lost, and the database is left valid, as when any scan dies (see -"Database"). A `SIGINT` that `scan` inherits as ignored, as a script's -background job does, stays ignored. +After the first signal `scan` stops catching them, so a second one ends it at +once, as an uncaught signal does: the records not yet committed are lost, and +the database is left valid, as when any scan dies (see "Database"). A `SIGINT` +that `scan` inherits as ignored, as a script's background job does, stays +ignored. ## Entrypoints This repository adheres to the [Scripts to Rule Them All](https://github.com/github/scripts-to-rule-them-all) -standard: the normalized executables in `script/` are the entrypoints -for the development workflow, and the `Makefile` targets are thin -shims that call them. Every script is POSIX `sh`, resolves the -repository root itself so it can be run from any working directory, -and may be invoked directly. The provided entrypoints are: +standard: the normalized executables in `script/` are the entrypoints for the +development workflow, and the `Makefile` targets are thin shims that call them. +Every script is POSIX `sh`, resolves the repository root itself so it can be run +from any working directory, and may be invoked directly. The provided +entrypoints are: -- `script/bootstrap` — install everything needed to build and - develop this repository, idempotently, assuming nothing is - present. `git`, `make`, and `go` come from the first of nix, apt, - brew, or apk found on the host, and are presence-checked only. - `golangci-lint` is deliberately **not** installed: it runs from a - digest-pinned image via `script/lint` and never from a host - install, so there is no host copy to drift from the pin. A missing - `docker` is warned about rather than installed or treated as - fatal — everything except linting works without it. Ends with - `go mod download`. -- `script/setup` — make a fresh clone ready for development: runs - `script/bootstrap`, then `script/install-precommit`. -- `script/projectname` — print this project's name (`sfdupes`). - Scripts that need the name call it, so they stay identical across - repositories. -- `script/test` — run the test suite with a 30-second timeout and - coverage enabled, rerunning verbosely on failure so the logs show - which test failed. -- `script/lint` — run the linter. It builds `Dockerfile.lint`, which - copies the repository into the digest-pinned - `golangci/golangci-lint` image and runs - `golangci-lint config verify` and `golangci-lint run` as build - steps, so a successful build is a clean lint. The linter is never - run on the host, which makes a working `docker` the one - prerequisite for linting — and therefore for `make check` and the - pre-commit hook. Offline machines: the gate steps themselves make - no network calls. `golangci-lint run` does not, and neither does - `golangci-lint config verify` — it validates against a schema the - pinned binary embeds, measured under `--network none` to both - pass a valid config and reject an invalid one. The build around - them does. `Dockerfile.lint` runs `go mod download` before the - gates and this module has external dependencies, so a first lint - on a machine with a cold BuildKit cache reaches the network there - (as well as pulling the pinned image); under `--network none` it - fails at that step, before any gate. That layer sits above the - gates and stays cached, so once it is warm `script/lint` — and - with it `make check` — runs entirely offline, until `go.mod` or - `go.sum` changes and the download layer goes cold again. Because - the daemon only ever sees a build context, this works when the - docker daemon is remote and bind mounts are impossible. -- `script/fmt` — format the Go sources in place (`gofmt -s -w`). - Markdown is not formatted. -- `script/fmt-check` — the read-only counterpart of `script/fmt`: - prints any unformatted file and exits non-zero instead of writing. -- `script/check` — run `script/test`, `script/lint`, and - `script/fmt-check`, in that order. Modifies nothing. Needs - `docker`, because `script/lint` does. -- `script/docker` — build the Docker image, tagged with the name - from `script/projectname`. The `Dockerfile` runs the gates as - build steps, so this is also the check a developer or reviewer - runs by hand. -- `script/cibuild` — build the Docker image untagged. This is what - the Gitea workflow runs on push; because the gates run as build - steps, a successful build implies the repository is green. -- `script/precommit` — run by the git pre-commit hook: `go mod tidy` - must be a no-op (a resulting change to `go.mod` or `go.sum` fails - the commit), then `script/check`. -- `script/install-precommit` — install the git pre-commit hook that - runs `script/precommit`. The hook is written to the common git - directory, so the main checkout and every worktree share it. -- `script/verify-lint-image-pin` — fail unless the - `golangci/golangci-lint` reference in `Dockerfile.lint` and the - one in the `Dockerfile` lint stage are the same image at the same - digest, naming both if not. The linter is pinned in those two - files and nothing else keeps them in sync, so a bump applied to - one alone would leave `make lint` and the `Dockerfile`'s - fail-fast lint stage checking the same tree against different - rulesets, both green. The guard restates neither pin — a third - copy would be the same drift one file further out — and runs as a - gate in both files, so `make lint`, `make check` and `make docker` - all catch it. +- `script/bootstrap` — install everything needed to build and develop this + repository, idempotently, assuming nothing is present. `git`, `make`, and `go` + come from the first of nix, apt, brew, or apk found on the host, and are + presence-checked only. `golangci-lint` and prettier are deliberately **not** + installed: they run in Docker (see `script/lint` and `script/fmt`) and never + from a host install, so there is no host copy to drift from the pin. A missing + `docker` is warned about rather than installed or treated as fatal — + everything except linting and formatting works without it. Ends with + `go mod download`. +- `script/setup` — make a fresh clone ready for development: runs + `script/bootstrap`, then `script/install-precommit`. +- `script/projectname` — print this project's name (`sfdupes`). Scripts that + need the name call it, so they stay identical across repositories. +- `script/test` — run the test suite with a 30-second timeout and coverage + enabled, rerunning verbosely on failure so the logs show which test failed. +- `script/lint` — run the linter. It builds `Dockerfile.lint`, which copies the + repository into the digest-pinned `golangci/golangci-lint` image and runs + `golangci-lint config verify` and `golangci-lint run` as build steps, so a + successful build is a clean lint. The linter is never run on the host, which + makes a working `docker` the one prerequisite for linting — and therefore for + `make check` and the pre-commit hook. Offline machines: the gate steps + themselves make no network calls. `golangci-lint run` does not, and neither + does `golangci-lint config verify` — it validates against a schema the pinned + binary embeds, measured under `--network none` to both pass a valid config and + reject an invalid one. The build around them does. `Dockerfile.lint` runs + `go mod download` before the gates and this module has external dependencies, + so a first lint on a machine with a cold BuildKit cache reaches the network + there (as well as pulling the pinned image); under `--network none` it fails + at that step, before any gate. That layer sits above the gates and stays + cached, so once it is warm `script/lint` — and with it `make check` — runs + entirely offline, until `go.mod` or `go.sum` changes and the download layer + goes cold again. Because the daemon only ever sees a build context, this works + when the docker daemon is remote and bind mounts are impossible. +- `script/fmt` — format in place: the Go sources with `gofmt -s -w`, and every + Markdown file with prettier, at the settings in `.prettierrc` (4-space + indents, prose wrapped at 80 columns). prettier is pinned by hash through + `package.json` and `yarn.lock` and never installed on the host: this builds + the `Dockerfile`'s `prettier` stage, a digest-pinned node image into which + `yarn install --frozen-lockfile` installs it, tagged `sfdupes-prettier`, and + runs that with the repository mounted, as the calling user. Needs `docker`, + and because of the mount, unlike `script/lint`, a local docker daemon. +- `script/fmt-check` — the read-only counterpart of `script/fmt`, with the + repository mounted read-only: prints any unformatted file and exits non-zero + instead of writing. gofmt and prettier both run every time, and each names + itself when it fails. The `Dockerfile` runs the same two checks as gates: the + gofmt check in its lint stage, prettier in its `markdown` stage. +- `script/check` — run `script/test`, `script/lint`, and `script/fmt-check`, in + that order. Modifies nothing. Needs `docker`, because `script/lint` and + `script/fmt-check` do. +- `script/docker` — build the Docker image, tagged with the name from + `script/projectname`. The `Dockerfile` runs the gates as build steps, so this + is also the check a developer or reviewer runs by hand. +- `script/cibuild` — build the Docker image untagged. This is what the Gitea + workflow runs on push; because the gates run as build steps, a successful + build implies the repository is green. +- `script/precommit` — run by the git pre-commit hook: `go mod tidy` must be a + no-op (a resulting change to `go.mod` or `go.sum` fails the commit), then + `script/check`. +- `script/install-precommit` — install the git pre-commit hook that runs + `script/precommit`. The hook is written to the common git directory, so the + main checkout and every worktree share it. +- `script/verify-lint-image-pin` — fail unless the `golangci/golangci-lint` + reference in `Dockerfile.lint` and the one in the `Dockerfile` lint stage are + the same image at the same digest, naming both if not. The linter is pinned in + those two files and nothing else keeps them in sync, so a bump applied to one + alone would leave `make lint` and the `Dockerfile`'s fail-fast lint stage + checking the same tree against different rulesets, both green. The guard + restates neither pin — a third copy would be the same drift one file further + out — and runs as a gate in both files, so `make lint`, `make check` and + `make docker` all catch it. -`script/verify-linter-pin` used to live here. It compared a linter -binary against a version pin in `script/bootstrap`, and both of its -subjects are gone: no linter binary is copied between build stages any -more, and bootstrap pins no version because it installs no linter. The -drift it existed to catch has moved from binary-versus-pin to -pin-versus-pin, which is what `script/verify-lint-image-pin` above -checks. +`script/verify-linter-pin` used to live here. It compared a linter binary +against a version pin in `script/bootstrap`, and both of its subjects are gone: +no linter binary is copied between build stages any more, and bootstrap pins no +version because it installs no linter. The drift it existed to catch has moved +from binary-versus-pin to pin-versus-pin, which is what +`script/verify-lint-image-pin` above checks. -`script/lint`, `script/docker` and `script/cibuild` all pass a freshly -computed `CHECK_EPOCH` build argument, and the gate steps in -`Dockerfile.lint` and `Dockerfile` reference it. Without that, an -unchanged tree lets Docker serve the gate layers from cache and the -build exits 0 having executed no tests and no lint — a green it never -earned, and one this repository has produced twice. `CHECK_EPOCH` -invalidates the gate layers on every run while leaving the pinned base -images and the dependency layers cached. `script/lint`'s value carries -the process id as well as the epoch, because two lint runs land inside -the same second easily and a bare epoch would cache the second one. +`script/lint`, `script/docker` and `script/cibuild` all pass a freshly computed +`CHECK_EPOCH` build argument, and the gate steps in `Dockerfile.lint` and +`Dockerfile` reference it. Without that, an unchanged tree lets Docker serve the +gate layers from cache and the build exits 0 having executed no tests and no +lint — a green it never earned, and one this repository has produced twice. +`CHECK_EPOCH` invalidates the gate layers on every run while leaving the pinned +base images and the dependency layers cached. `script/lint`'s value carries the +process id as well as the epoch, because two lint runs land inside the same +second easily and a bare epoch would cache the second one. ## Build -The `script/` entrypoints above are where the implementations live; -the `Makefile` targets are shims onto them, except `build`, which -carries the compile recipe: +The `script/` entrypoints above are where the implementations live; the +`Makefile` targets are shims onto them, except `build`, which carries the +compile recipe: -- `make` / `make build` — build the `sfdupes` binary (cgo - disabled); building is the default target. -- `make bootstrap` — install the build and development - dependencies. -- `make setup` — prepare a fresh clone: `bootstrap` plus the - pre-commit hook. -- `make test` — run the test suite (30-second timeout; reruns with - `-v` on failure). -- `make lint` — run `golangci-lint` with the repo config, in Docker - (see `script/lint`); requires `docker`. -- `make fmt` / `make fmt-check` — format Go sources / verify - formatting without writing. -- `make check` — `test`, `lint`, and `fmt-check`; modifies nothing. - Requires `docker`, via `lint`. -- `make docker` — build the Docker image, which runs the gates as - build stages. -- `make hooks` — install the pre-commit hook. -- `make clean` — remove the binary. +- `make` / `make build` — build the `sfdupes` binary (cgo disabled); building is + the default target. +- `make bootstrap` — install the build and development dependencies. +- `make setup` — prepare a fresh clone: `bootstrap` plus the pre-commit hook. +- `make test` — run the test suite (30-second timeout; reruns with `-v` on + failure). +- `make lint` — run `golangci-lint` with the repo config, in Docker (see + `script/lint`); requires `docker`. +- `make fmt` / `make fmt-check` — format the Go sources and the Markdown / + verify formatting without writing; requires `docker`, for prettier (see + `script/fmt`). +- `make check` — `test`, `lint`, and `fmt-check`; modifies nothing. Requires + `docker`, via `lint` and `fmt-check`. +- `make docker` — build the Docker image, which runs the gates as build stages. +- `make hooks` — install the pre-commit hook. +- `make clean` — remove the binary. ### Definition of done All of the following, run in this directory, must pass: -1. `make check` passes (tests, lint, `gofmt`). +1. `make check` passes (tests, lint, `gofmt`, prettier). 2. `make docker` succeeds. -3. Smoke test — create a throwaway tree in a temp dir (never test - against real data): +3. Smoke test — create a throwaway tree in a temp dir (never test against real + data): ```sh d=$(mktemp -d) @@ -962,30 +883,30 @@ All of the following, run in this directory, must pass: ./sfdupes report ``` - (The database lives in a temp directory of its own: inside `$d`, - the scan would record it, and its empty lock file would join the - `empty1`/`empty2` group.) + (The database lives in a temp directory of its own: inside `$d`, the scan + would record it, and its empty lock file would join the `empty1`/`empty2` + group.) - Expected from the first `report`: `one.bin`/`copy.bin`/`copy2.bin` - form one group (two dupe rows, `first` is the lexicographically - smallest path); `t1/f1`/`t2/f1`/`t3/f1` form one group; - `t1/sub/f2`/ `t2/sub/f2`/`t3/sub/f2renamed` form one group; - `tiny1`/`tiny2` pair; `empty1`/`empty2` pair; `unique.bin` and - `tiny3` appear nowhere; groups ordered by size descending. + Expected from the first `report`: `one.bin`/`copy.bin`/`copy2.bin` form one + group (two dupe rows, `first` is the lexicographically smallest path); + `t1/f1`/`t2/f1`/`t3/f1` form one group; `t1/sub/f2`/ + `t2/sub/f2`/`t3/sub/f2renamed` form one group; `tiny1`/`tiny2` pair; + `empty1`/`empty2` pair; `unique.bin` and `tiny3` appear nowhere; groups + ordered by size descending. - Expected from `trees`: exactly one row — `first` `$d/t1`, `dupe` - `$d/t2`, 2 files, 3100 bytes. `$d/t1/sub` vs `$d/t2/sub` is - suppressed as non-maximal (implied by the `t1`/`t2` group), and `t3` - appears nowhere (its file set differs by name). + Expected from `trees`: exactly one row — `first` `$d/t1`, `dupe` `$d/t2`, 2 + files, 3100 bytes. `$d/t1/sub` vs `$d/t2/sub` is suppressed as non-maximal + (implied by the `t1`/`t2` group), and `t3` appears nowhere (its file set + differs by name). Expected from the second `report` (after the modify/delete rescan): `one.bin` has left its group (its content changed), so - `copy.bin`/`copy2.bin` remain as one pair, and `unique.bin` is - gone from the database. + `copy.bin`/`copy2.bin` remain as one pair, and `unique.bin` is gone from the + database. - The test suite automates this scenario (see `scan_test.go`), plus a - negative check: `report` and `trees` operate on the database alone - and never touch the scanned filesystem. + The test suite automates this scenario (see `scan_test.go`), plus a negative + check: `report` and `trees` operate on the database alone and never touch + the scanned filesystem. ## TODO @@ -993,13 +914,13 @@ Tracked in [TODO.md](TODO.md). ## Non-goals -- No byte-for-byte compare, and no deletion or linking of - duplicates. Files that match are compared by a SHA-256 of the whole - file below 50 MiB, and only by samples at 50 MiB and over. The - reports are advisory; acting on them is the user's job. -- No persistence beyond the SQLite database described above; no - export/import formats. -- No daemon or filesystem watcher; scheduling rescans is cron's job. +- No byte-for-byte compare, and no deletion or linking of duplicates. Files that + match are compared by a SHA-256 of the whole file below 50 MiB, and only by + samples at 50 MiB and over. The reports are advisory; acting on them is the + user's job. +- No persistence beyond the SQLite database described above; no export/import + formats. +- No daemon or filesystem watcher; scheduling rescans is cron's job. ## License diff --git a/TODO.md b/TODO.md index b465821..789bc67 100644 --- a/TODO.md +++ b/TODO.md @@ -1,43 +1,45 @@ # Workflow -- take an issue from the `1.0.0` milestone on the tracker; work not - yet on the tracker gets filed as an issue first +- take an issue from the `1.0.0` milestone on the tracker; work not yet on the + tracker gets filed as an issue first - branch (from `main`) - do the work, with tests, in small focused commits -- record it at the top of Completed Steps (`TODO.md` changes in the - same commit as the work) -- push the branch and open a PR whose title ends with - ` (closes #N)` -- an independent review gates the merge; every finding is addressed - or explicitly rebutted on the PR +- record it at the top of Completed Steps (`TODO.md` changes in the same commit + as the work) +- push the branch and open a PR whose title ends with ` (closes #N)` +- an independent review gates the merge; every finding is addressed or + explicitly rebutted on the PR - merge to `main` once the review passes # Status - pre-1.0 -- the Gitea tracker is authoritative for the pre-1.0 backlog: the - open issues under the `1.0.0` milestone are what remains before - the tag, and this file records history and process, not the queue +- the Gitea tracker is authoritative for the pre-1.0 backlog: the open issues + under the `1.0.0` milestone are what remains before the tag, and this file + records history and process, not the queue # Next Step - take the next issue from the `1.0.0` milestone on the tracker: - https://git.eeqj.de/sneak/sfdupes/milestone/17 — the milestone is - the source of truth for what is left before 1.0.0. Individual - issues are deliberately not restated here; a copy in this file - drifts out of date the moment the tracker moves + https://git.eeqj.de/sneak/sfdupes/milestone/17 — the milestone is the source + of truth for what is left before 1.0.0. Individual issues are deliberately not + restated here; a copy in this file drifts out of date the moment the tracker + moves # Completed Steps -- `scan` rejects `--workers` below 1 as a usage error instead of - running single-threaded (2026-10-04, - https://git.eeqj.de/sneak/sfdupes/issues/10) +- `make fmt` and `make fmt-check` run prettier over all Markdown, in Docker, and + CI checks it; all Markdown reformatted (2026-10-04, + https://git.eeqj.de/sneak/sfdupes/issues/19) -- a test fails when either walk cancellation check in `scan.go` is - removed (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/81) +- `scan` rejects `--workers` below 1 as a usage error instead of running + single-threaded (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/10) -- test that `scan` refuses a database with another schema version - (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/64) +- a test fails when either walk cancellation check in `scan.go` is removed + (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/81) + +- test that `scan` refuses a database with another schema version (2026-10-04, + https://git.eeqj.de/sneak/sfdupes/issues/64) - correct four inaccurate comments in `cancel_test.go` and rename `walkCancelInFlightDirs` to `walkCancelInFlightFiles` (2026-10-04, @@ -50,479 +52,428 @@ `files` table is refused with a clear schema-version error (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/11) -- README documents install, Docker, a daily cron scan and how to read - and check the reports (2026-10-04, - https://git.eeqj.de/sneak/sfdupes/issues/54) +- README documents install, Docker, a daily cron scan and how to read and check + the reports (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/54) -- the `Dockerfile` build stage keeps the Go module cache out of `builder`'s - home and copies the sources with `--chown`, so no `chown -R` walks them +- the `Dockerfile` build stage keeps the Go module cache out of `builder`'s home + and copies the sources with `--chown`, so no `chown -R` walks them (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/43) - `--version` prints `sfdupes VERSION` to stdout; README documents it and `--help` (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/15) -- `scan` stops cleanly on `SIGINT` or `SIGTERM`: commits what it has - hashed, deletes nothing more, exits 1 (2026-10-04, +- `scan` stops cleanly on `SIGINT` or `SIGTERM`: commits what it has hashed, + deletes nothing more, exits 1 (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/5) -- `report` and `trees` stream the records instead of holding them all in - memory; the schema gains the `files_signature` index (2026-10-04, +- `report` and `trees` stream the records instead of holding them all in memory; + the schema gains the `files_signature` index (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/14) -- progress prints at once on a non-terminal, uses a real terminal test, - and prints warnings through a spinner instead of racing its redraw - (2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/13) +- progress prints at once on a non-terminal, uses a real terminal test, and + prints warnings through a spinner instead of racing its redraw (2026-10-03, + https://git.eeqj.de/sneak/sfdupes/issues/13) -- warn about and skip symlink, socket, FIFO, device and `.zfs` - operands, keeping the records beneath them (2026-10-03, +- warn about and skip symlink, socket, FIFO, device and `.zfs` operands, keeping + the records beneath them (2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/9) -- `scan` holds a lock on a lock file beside the database for its whole run, - so a second `scan` fails at once with exit 1 (2026-10-03, +- `scan` holds a lock on a lock file beside the database for its whole run, so a + second `scan` fails at once with exit 1 (2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/53) - test stdout write failures in `report` and `trees`; README states that `| head` ends sfdupes by `SIGPIPE` and `>&-` writes to `/dev/null` (2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/30) -- `report` and `trees` open the database read-only, and `scan` leaves it - out of WAL mode, so reading needs only read access (2026-10-03, closes +- `report` and `trees` open the database read-only, and `scan` leaves it out of + WAL mode, so reading needs only read access (2026-10-03, closes https://git.eeqj.de/sneak/sfdupes/issues/8) -- escape tabs, newlines, carriage returns and backslashes in report, - trees and warning paths; the root directory's path is `/` - (2026-10-03, https://git.eeqj.de/sneak/sfdupes/issues/7) +- escape tabs, newlines, carriage returns and backslashes in report, trees and + warning paths; the root directory's path is `/` (2026-10-03, + https://git.eeqj.de/sneak/sfdupes/issues/7) -- stamp the git tag or short commit in a plain `docker build .` - instead of `dev` (2026-10-02, branch `next`, closes - https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now - sends `.git`, without `.git/config`, and the `Dockerfile` build - stage takes the `VERSION` build argument when one is given, - otherwise `git describe --tags --always` of that `.git`. The build - fails if the context carries `.git` and the version still comes out - empty, `dev` or `unknown`. The CI checkout step fetches the full - history (`fetch-depth: 0`) so CI sees the tag and stamps the same - value as `make build`. +- stamp the git tag or short commit in a plain `docker build .` instead of `dev` + (2026-10-02, branch `next`, closes + https://git.eeqj.de/sneak/sfdupes/issues/67): `.dockerignore` now sends + `.git`, without `.git/config`, and the `Dockerfile` build stage takes the + `VERSION` build argument when one is given, otherwise + `git describe --tags --always` of that `.git`. The build fails if the context + carries `.git` and the version still comes out empty, `dev` or `unknown`. The + CI checkout step fetches the full history (`fetch-depth: 0`) so CI sees the + tag and stamps the same value as `make build`. -- replace the 1 KiB end-window sampling with the head/tail plus - content-hash ladder (2026-09-22, branch `next`, closes - https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is - hashed in full and compared directly, with no end-window step — its - `head`, `tail`, and `content` all hold the whole-file hash. A file at - 10 MiB or above gets only the 64 KiB `head` and `tail` in the hash - phase; a new content phase, after the update phase, reads it for its - `content` hash — the whole file below 50 MiB, gigabyte-spaced 1 MiB - samples at or above — only when its size, `head`, and `tail` match - another record's, from the same scan or stored by an earlier one, so - a stored file gains its content hash when it gains a match. A file - that is gone or has changed since its record was written is not - read. The `content` column is part of the version 1 schema. `report` - and `trees` group by the extended signature and leave out any record - without a `content` hash, so the ladder is applied across the whole - database. README "Duplicate detection" documents every rung including - the probabilistic large-file path. +- replace the 1 KiB end-window sampling with the head/tail plus content-hash + ladder (2026-09-22, branch `next`, closes + https://git.eeqj.de/sneak/sfdupes/issues/61): a file under 10 MiB is hashed in + full and compared directly, with no end-window step — its `head`, `tail`, and + `content` all hold the whole-file hash. A file at 10 MiB or above gets only + the 64 KiB `head` and `tail` in the hash phase; a new content phase, after the + update phase, reads it for its `content` hash — the whole file below 50 MiB, + gigabyte-spaced 1 MiB samples at or above — only when its size, `head`, and + `tail` match another record's, from the same scan or stored by an earlier one, + so a stored file gains its content hash when it gains a match. A file that is + gone or has changed since its record was written is not read. The `content` + column is part of the version 1 schema. `report` and `trees` group by the + extended signature and leave out any record without a `content` hash, so the + ladder is applied across the whole database. README "Duplicate detection" + documents every rung including the probabilistic large-file path. -- remove the dead `files.dat` references from `Makefile`, `.gitignore` - and `.dockerignore` (2026-09-21, branch `next`, closes +- remove the dead `files.dat` references from `Makefile`, `.gitignore` and + `.dockerignore` (2026-09-21, branch `next`, closes https://git.eeqj.de/sneak/sfdupes/issues/22) - fix the lint-image pin comments and `FROM` form in `Dockerfile` and `Dockerfile.lint` (2026-08-10, branch `next`, closes https://git.eeqj.de/sneak/sfdupes/issues/25): dropped the false - `(Debian-based)` parenthetical (v2.12.1 was Debian too) and the - redundant tag, so both pins are the policy `# image:vX.Y.Z, - YYYY-MM-DD` comment over a bare `FROM image@sha256:...`. Digest - unchanged. `script/verify-lint-image-pin` parses those `FROM` lines - and still matches the tagless form; its advice line lost the now - meaningless "tag and digest". With no tag in either reference, a - tag-only disagreement no longer exists — a one-sided tag is caught as - a plain mismatch. + `(Debian-based)` parenthetical (v2.12.1 was Debian too) and the redundant tag, + so both pins are the policy `# image:vX.Y.Z, YYYY-MM-DD` comment over a bare + `FROM image@sha256:...`. Digest unchanged. `script/verify-lint-image-pin` + parses those `FROM` lines and still matches the tagless form; its advice line + lost the now meaningless "tag and digest". With no tag in either reference, a + tag-only disagreement no longer exists — a one-sided tag is caught as a plain + mismatch. -- run all linting in Docker via `Dockerfile.lint` and `script/lint` - (2026-08-10, branch `next`, closes - https://git.eeqj.de/sneak/sfdupes/issues/46): per the owner ruling, the - linter runs inside a container invoked through the `script/` - entrypoint and is never installed on a host. New root - `Dockerfile.lint` COPYs the repo into the digest-pinned - `golangci/golangci-lint:v2.12.2` image and runs - `golangci-lint config verify` and `golangci-lint run` as build - steps, so a successful build IS a clean lint; `script/lint` is - reduced to building it. `script/bootstrap` loses the `go install`, - the pin constants, the version parser and `verify_golangci_lint` - outright rather than hardening them — with nothing linting on the - host, the `$GOPATH/bin` versus `PATH` problem that motivated them has - no subject — and now warns rather than fails when `docker` is absent. - Two traps handled. A lint build on an unchanged tree returns success - in well under a second having run no linter, which is +- run all linting in Docker via `Dockerfile.lint` and `script/lint` (2026-08-10, + branch `next`, closes https://git.eeqj.de/sneak/sfdupes/issues/46): per the + owner ruling, the linter runs inside a container invoked through the `script/` + entrypoint and is never installed on a host. New root `Dockerfile.lint` COPYs + the repo into the digest-pinned `golangci/golangci-lint:v2.12.2` image and + runs `golangci-lint config verify` and `golangci-lint run` as build steps, so + a successful build IS a clean lint; `script/lint` is reduced to building it. + `script/bootstrap` loses the `go install`, the pin constants, the version + parser and `verify_golangci_lint` outright rather than hardening them — with + nothing linting on the host, the `$GOPATH/bin` versus `PATH` problem that + motivated them has no subject — and now warns rather than fails when `docker` + is absent. Two traps handled. A lint build on an unchanged tree returns + success in well under a second having run no linter, which is https://git.eeqj.de/sneak/sfdupes/issues/32 and - https://git.eeqj.de/sneak/sfdupes/issues/39 again, so - `Dockerfile.lint` carries `ARG CHECK_EPOCH` referenced - inside every gate `RUN` (BuildKit hashes the expanded command, not - the declaration) and `script/lint` passes `"$(date +%s)-$$"` — the - PID matters because two lint runs land inside the same second easily. - And nothing inside an image build may shell out to docker, so the - main `Dockerfile`'s lint stage now invokes `golangci-lint` directly + https://git.eeqj.de/sneak/sfdupes/issues/39 again, so `Dockerfile.lint` + carries `ARG CHECK_EPOCH` referenced inside every gate `RUN` (BuildKit hashes + the expanded command, not the declaration) and `script/lint` passes + `"$(date +%s)-$$"` — the PID matters because two lint runs land inside the + same second easily. And nothing inside an image build may shell out to docker, + so the main `Dockerfile`'s lint stage now invokes `golangci-lint` directly instead of `make lint`, and its build stage runs `make test` and - `make fmt-check` instead of the `make check` aggregate (`make`, not - the scripts bare, because the Makefile's `export CGO_ENABLED = 0` - only reaches what it invokes). `COPY --from=lint` - `/usr/bin/golangci-lint` is replaced by - `COPY --from=lint /src/go.sum /dev/null`: the copied binary was the - only edge forcing BuildKit to finish linting before the build stage - starts, and dropping it without replacing the edge would have ended - fail-fast linting silently under a still-green build. That is - canonical `REPO_POLICIES.md:107`'s ordering edge, restored. - `ENV PATH=/home/builder/go/bin:$PATH` is gone with the `go install` - that justified it. `script/verify-linter-pin` is retired, deleted - along with its README entry, because both of its subjects ceased to - exist in the same change: it compared a linter binary against - `GOLANGCI_LINT_VERSION` in `script/bootstrap`, and there is now - neither a binary crossing between stages nor a version pin in - bootstrap. The drift it guarded has not gone away, it has moved — the - linter is still pinned twice, now as the `FROM` line of - `Dockerfile.lint` and the `FROM` line of the `Dockerfile` lint stage, - with nothing syncing them, which is exactly what + `make fmt-check` instead of the `make check` aggregate (`make`, not the + scripts bare, because the Makefile's `export CGO_ENABLED = 0` only reaches + what it invokes). `COPY --from=lint` `/usr/bin/golangci-lint` is replaced by + `COPY --from=lint /src/go.sum /dev/null`: the copied binary was the only edge + forcing BuildKit to finish linting before the build stage starts, and dropping + it without replacing the edge would have ended fail-fast linting silently + under a still-green build. That is canonical `REPO_POLICIES.md:107`'s ordering + edge, restored. `ENV PATH=/home/builder/go/bin:$PATH` is gone with the + `go install` that justified it. `script/verify-linter-pin` is retired, deleted + along with its README entry, because both of its subjects ceased to exist in + the same change: it compared a linter binary against `GOLANGCI_LINT_VERSION` + in `script/bootstrap`, and there is now neither a binary crossing between + stages nor a version pin in bootstrap. The drift it guarded has not gone away, + it has moved — the linter is still pinned twice, now as the `FROM` line of + `Dockerfile.lint` and the `FROM` line of the `Dockerfile` lint stage, with + nothing syncing them, which is exactly what https://git.eeqj.de/sneak/sfdupes/issues/42 made a build failure. Its - replacement is one new `script/verify-lint-image-pin`, - run as a gate in both files, which compares the two references to - each other and deliberately restates neither: a hardcoded expected - digest would be a third copy and the same drift one file further out. - `golangci-lint config verify` is included per the ruling, and the - concern about its unpinned live HTTPS schema fetch was measured - rather than assumed — under `--network none` the pinned binary both - passes a valid config and rejects an invalid one with the jsonschema - error, so it validates from an embedded schema and makes no network - call of its own. The README scopes that to the gate steps rather - than to linting as a whole: `Dockerfile.lint` runs `go mod download` - above them, so a cold cache still needs the network and only a warm - one lints offline. Verified: `make lint` green with every `PATH` - directory containing a `golangci-lint` removed + replacement is one new `script/verify-lint-image-pin`, run as a gate in both + files, which compares the two references to each other and deliberately + restates neither: a hardcoded expected digest would be a third copy and the + same drift one file further out. `golangci-lint config verify` is included per + the ruling, and the concern about its unpinned live HTTPS schema fetch was + measured rather than assumed — under `--network none` the pinned binary both + passes a valid config and rejects an invalid one with the jsonschema error, so + it validates from an embedded schema and makes no network call of its own. The + README scopes that to the gate steps rather than to linting as a whole: + `Dockerfile.lint` runs `go mod download` above them, so a cold cache still + needs the network and only a warm one lints offline. Verified: `make lint` + green with every `PATH` directory containing a `golangci-lint` removed (`/home/user/go/bin`, `/home/user/.local/bin`, `/usr/local/bin`; - `command -v golangci-lint` empty); two consecutive `script/lint` runs - on an untouched tree both executed the linter, 27.7s and 28.7s in the - lint step under distinct epochs with the `COPY . .` layer `CACHED` - above them, at 42.2s and 41.8s wall clock — the no-cache rule was not - weakened to shorten that. Negative control: a planted - `var unusedIssue46Sentinel = 1` failed `script/lint` with - `report.go:173:5: var unusedIssue46Sentinel is unused (unused)`, and - failed `make docker` at `[lint 9/9]` with the build stage stopped at - `[builder 3/12]` — `COPY --from=lint`, `script/bootstrap`, the test - gate and `make build` all zero occurrences — then reverted clean. The - drift guard fails on a tag-only disagreement, on a digest-only - disagreement, and on an unreadable reference, naming both sides. - `make docker` green in 5m35s with all six gates executing under one - epoch (lint 37.6s, test 25.2s reporting - `ok sneak.berlin/go/sfdupes 1.938s coverage: 88.5%`, not `(cached)`). - The non-root quirk still holds: in the builder image with the Go test - cache off, `--user 0:0` fails `TestScanHardlinkRunFailsTogether` - (exit 1) where the unprivileged user passes (exit 0). Noted for - follow-up, not fixed here: `golangci-lint` warns that the - `gomodguard` linter is deprecated since v2.12.0 in favour of - `gomodguard_v2`. + `command -v golangci-lint` empty); two consecutive `script/lint` runs on an + untouched tree both executed the linter, 27.7s and 28.7s in the lint step + under distinct epochs with the `COPY . .` layer `CACHED` above them, at 42.2s + and 41.8s wall clock — the no-cache rule was not weakened to shorten that. + Negative control: a planted `var unusedIssue46Sentinel = 1` failed + `script/lint` with + `report.go:173:5: var unusedIssue46Sentinel is unused (unused)`, and failed + `make docker` at `[lint 9/9]` with the build stage stopped at `[builder 3/12]` + — `COPY --from=lint`, `script/bootstrap`, the test gate and `make build` all + zero occurrences — then reverted clean. The drift guard fails on a tag-only + disagreement, on a digest-only disagreement, and on an unreadable reference, + naming both sides. `make docker` green in 5m35s with all six gates executing + under one epoch (lint 37.6s, test 25.2s reporting + `ok sneak.berlin/go/sfdupes 1.938s coverage: 88.5%`, not `(cached)`). The + non-root quirk still holds: in the builder image with the Go test cache off, + `--user 0:0` fails `TestScanHardlinkRunFailsTogether` (exit 1) where the + unprivileged user passes (exit 0). Noted for follow-up, not fixed here: + `golangci-lint` warns that the `gomodguard` linter is deprecated since v2.12.0 + in favour of `gomodguard_v2`. -- install the Docker build stage's prerequisites by running - `script/bootstrap` instead of `apk add --no-cache make` inline - (2026-08-09, branch `dockerfile-bootstrap`, closes #42): canonical - `REPO_POLICIES.md:97` requires it, and the inline install left the - build stage maintaining its own notion of the toolchain — exactly - the divergence #24 exists to close, one layer down. The stage now - copies `script/` plus `go.mod`/`go.sum` and runs `script/bootstrap`, - which ends in `go mod download`, so the separate invocation of that - is gone. `COPY --from=lint /usr/bin/golangci-lint` stays, and moves - above the bootstrap layer. It is the only edge making this stage - depend on the lint stage, so deleting it as redundant would end - fail-fast linting silently. Letting bootstrap install its own linter - here would have reintroduced the second toolchain and paid for a - from-source build of it. What makes the two stages provably one - toolchain rather than two that happen to agree is a new - `script/verify-linter-pin`, run in the build stage on the binary - that arrives from the lint stage, before bootstrap: it fails the - build naming both versions unless that binary is the version - `script/bootstrap` pins. Bootstrap's own check could not serve that - purpose — it reinstalls its pin from source and then verifies - whatever `PATH` resolves, so drift self-heals silently and a lint - stage image bumped on its own would lint at the new version while - `make check` ran at the old one, green. The linter version is pinned - in two independent places (the lint stage image digest and +- install the Docker build stage's prerequisites by running `script/bootstrap` + instead of `apk add --no-cache make` inline (2026-08-09, branch + `dockerfile-bootstrap`, closes #42): canonical `REPO_POLICIES.md:97` requires + it, and the inline install left the build stage maintaining its own notion of + the toolchain — exactly the divergence #24 exists to close, one layer down. + The stage now copies `script/` plus `go.mod`/`go.sum` and runs + `script/bootstrap`, which ends in `go mod download`, so the separate + invocation of that is gone. `COPY --from=lint /usr/bin/golangci-lint` stays, + and moves above the bootstrap layer. It is the only edge making this stage + depend on the lint stage, so deleting it as redundant would end fail-fast + linting silently. Letting bootstrap install its own linter here would have + reintroduced the second toolchain and paid for a from-source build of it. What + makes the two stages provably one toolchain rather than two that happen to + agree is a new `script/verify-linter-pin`, run in the build stage on the + binary that arrives from the lint stage, before bootstrap: it fails the build + naming both versions unless that binary is the version `script/bootstrap` + pins. Bootstrap's own check could not serve that purpose — it reinstalls its + pin from source and then verifies whatever `PATH` resolves, so drift + self-heals silently and a lint stage image bumped on its own would lint at the + new version while `make check` ran at the old one, green. The linter version + is pinned in two independent places (the lint stage image digest and `GOLANGCI_LINT_VERSION`) and nothing else keeps them in sync, so a half-applied bump is now a build failure. The pin is read out of - `script/bootstrap`, which stays the single source of truth; a pin - that cannot be read is a hard failure, not a skip. The check needs - no `CHECK_EPOCH`: its only inputs are the copied binary and - `script/`, so Docker invalidates the layer exactly when a cached - result would stop being true, and it is documented with the other - entrypoints in the README. `$GOPATH/bin` joins `PATH` because - that is where bootstrap's `go install` lands and bootstrap verifies - its installs against what `PATH` resolves — nothing in the image is - shadowed by it, the directory does not exist until bootstrap runs. - Everything added sits above `ARG CHECK_EPOCH`, and the `chown` and - `USER builder` still precede `make check`. Verified: the guard fails - the build with both versions named when the lint stage's linter is - faked to a different version, and an unmodified build still passes - it; bootstrap runs clean under Alpine's `sh` and its `apk` branch, - installing `git` and `make` and finding the copied - linter already at the pin; a second build served the bootstrap and - dependency layers `CACHED` while both gates ran with a fresh epoch; - a planted `unused` finding failed the build at the lint gate in - 48.9s with the build stage's `make check` never starting; and the - suite run in the image as `--user 0:0` fails - `TestScanHardlinkRunFailsTogether`, so the drop to the unprivileged - user is still load-bearing. That last check needs the Go test cache - disabled — the first attempt reported `ok ... (cached)` as root, - reusing the result the build-time run had left in the shared cache, - which would have read as a pass. Build wall time, on a shared host - running many concurrent builds and so noisy: 2m13s on an unchanged - tree, 2m17s and 4m29s for two builds after a source change, 5m14s - cold. Only the cold one breaches the policy ceiling, and not because - of this change — `chown -R builder:builder /src /home/builder` walks - the module cache and re-runs on every source change, and it alone - varied between 77s and 210s across those four builds, which is also - the whole spread in the totals. The same cold measurement against - `main` is 5m03s with a 209s `chown`. Filed as #43 -- bust the Docker layer cache for the gate steps, so `script/cibuild` - and `script/docker` cannot report a green they did not earn - (2026-08-09, branch `cibuild-cache-bust`, closes #32): both scripts - were bare `docker build` invocations with no cache control, and the - `Dockerfile` copies the tree before running its gates, so on an - unchanged tree Docker served those layers from cache and the build - exited 0 having executed nothing. That is not hypothetical here — - every merge this repo has done is a non-fast-forward merge of an - undiverged branch, so each merge commit's tree is byte-identical to - the branch head's and each merge CI run was almost certainly a full - cache hit; and PR #31's reviewer found `make docker` returning - success as a 17-layer cache hit, catching it only by being - suspicious. The fix is `ARG CHECK_EPOCH` with the scripts passing - `--build-arg CHECK_EPOCH="$(date +%s)"`. Two details make or break - it. `ARG` is scoped per stage and this `Dockerfile` has three gates - across two — `make fmt-check` and `make lint` in the lint stage, - `make check` in the build stage — so a single declaration would have - left one stage silently cacheable; it is declared in both. And - BuildKit hashes the expanded command, not the declaration, so a - declared-but-unreferenced `ARG` invalidates nothing: each gate `RUN` - echoes the epoch, which also puts the value in the build log as - evidence the layer really ran. Placement is below the dependency - layers on purpose — a build that goes cold every time would be a - different bug, not a fix. Verified by running each script twice back - to back on an unchanged tree under `BUILDKIT_PROGRESS=plain`: all - three gates executed on all four runs, each with a fresh epoch in - the log (`script/cibuild` 78.8s then 61.1s; `script/docker` 61.1s - then 53.4s), and twelve steps were still served `CACHED` in the - steady state — both `go mod download`s, `apk add`, `adduser`, the - `chown`, every `go.mod`/`go.sum` and source copy, the linter copy - out of the lint stage, and the binary copy into the runtime stage. - The lint stage still gates the build stage: with a deliberate - `unused` finding planted in the tree, the build failed at - `make lint` in 36.1s and the build-stage `make check` never started. - The build stage also still drops to the unprivileged `builder` user - before `make check`, which the suite depends on rather than merely - prefers: forcing the same image to run the tests as root fails - `TestScanHardlinkRunFailsTogether`, because root reads straight - through the `chmod(0)` the test uses to prove hard links are read - once. This is the local fix only; propagating it to the canonical - templates is `prompts` #26 -- check the installed golangci-lint version in `script/bootstrap` - instead of only its presence (2026-08-09, branch - `bootstrap-version-check`, closes #24): `missing golangci-lint` meant - any linter already on `PATH` satisfied the check, so the pin was never - consulted and the v2.12.2 bump from #3 was inert on every host that - already had one — this host ran v2.10.1 against a v2.12.2 pin, - `make check` went green, and `make docker` then rejected the same - commit with findings the local gate never saw. The version now lives - in one place, `GOLANGCI_LINT_VERSION`, with the `go install` module - ref derived from it so a bump cannot half-apply; a - `golangci_lint_version` helper parses `golangci-lint --version` - (taking the field after the word `version` and tolerating an optional - leading `v`, which the module ref carries and the binary's output does - not), and any version that is not the pin — older, newer, absent or - unparseable — is reinstalled. The install is then verified against the - binary `PATH` actually resolves: `go install` writes into `GOBIN` (or - `GOPATH/bin`) while `make lint` runs whichever `golangci-lint` comes - first on `PATH`, so a wrong-version one sitting ahead of it — nix, - apt, brew, apk, or the `/usr/local/bin` copy the `Dockerfile` builder - stage makes — would swallow the install and leave the local gate - disagreeing with CI under an affirmative `bootstrap complete`. - Bootstrap now re-reads the effective version after installing and, on - a mismatch, prints both paths and both versions to stderr and exits - non-zero instead of claiming success; it does not reorder anyone's + `script/bootstrap`, which stays the single source of truth; a pin that cannot + be read is a hard failure, not a skip. The check needs no `CHECK_EPOCH`: its + only inputs are the copied binary and `script/`, so Docker invalidates the + layer exactly when a cached result would stop being true, and it is documented + with the other entrypoints in the README. `$GOPATH/bin` joins `PATH` because + that is where bootstrap's `go install` lands and bootstrap verifies its + installs against what `PATH` resolves — nothing in the image is shadowed by + it, the directory does not exist until bootstrap runs. Everything added sits + above `ARG CHECK_EPOCH`, and the `chown` and `USER builder` still precede + `make check`. Verified: the guard fails the build with both versions named + when the lint stage's linter is faked to a different version, and an + unmodified build still passes it; bootstrap runs clean under Alpine's `sh` and + its `apk` branch, installing `git` and `make` and finding the copied linter + already at the pin; a second build served the bootstrap and dependency layers + `CACHED` while both gates ran with a fresh epoch; a planted `unused` finding + failed the build at the lint gate in 48.9s with the build stage's `make check` + never starting; and the suite run in the image as `--user 0:0` fails + `TestScanHardlinkRunFailsTogether`, so the drop to the unprivileged user is + still load-bearing. That last check needs the Go test cache disabled — the + first attempt reported `ok ... (cached)` as root, reusing the result the + build-time run had left in the shared cache, which would have read as a pass. + Build wall time, on a shared host running many concurrent builds and so noisy: + 2m13s on an unchanged tree, 2m17s and 4m29s for two builds after a source + change, 5m14s cold. Only the cold one breaches the policy ceiling, and not + because of this change — `chown -R builder:builder /src /home/builder` walks + the module cache and re-runs on every source change, and it alone varied + between 77s and 210s across those four builds, which is also the whole spread + in the totals. The same cold measurement against `main` is 5m03s with a 209s + `chown`. Filed as #43 +- bust the Docker layer cache for the gate steps, so `script/cibuild` and + `script/docker` cannot report a green they did not earn (2026-08-09, branch + `cibuild-cache-bust`, closes #32): both scripts were bare `docker build` + invocations with no cache control, and the `Dockerfile` copies the tree before + running its gates, so on an unchanged tree Docker served those layers from + cache and the build exited 0 having executed nothing. That is not hypothetical + here — every merge this repo has done is a non-fast-forward merge of an + undiverged branch, so each merge commit's tree is byte-identical to the branch + head's and each merge CI run was almost certainly a full cache hit; and PR + #31's reviewer found `make docker` returning success as a 17-layer cache hit, + catching it only by being suspicious. The fix is `ARG CHECK_EPOCH` with the + scripts passing `--build-arg CHECK_EPOCH="$(date +%s)"`. Two details make or + break it. `ARG` is scoped per stage and this `Dockerfile` has three gates + across two — `make fmt-check` and `make lint` in the lint stage, `make check` + in the build stage — so a single declaration would have left one stage + silently cacheable; it is declared in both. And BuildKit hashes the expanded + command, not the declaration, so a declared-but-unreferenced `ARG` invalidates + nothing: each gate `RUN` echoes the epoch, which also puts the value in the + build log as evidence the layer really ran. Placement is below the dependency + layers on purpose — a build that goes cold every time would be a different + bug, not a fix. Verified by running each script twice back to back on an + unchanged tree under `BUILDKIT_PROGRESS=plain`: all three gates executed on + all four runs, each with a fresh epoch in the log (`script/cibuild` 78.8s then + 61.1s; `script/docker` 61.1s then 53.4s), and twelve steps were still served + `CACHED` in the steady state — both `go mod download`s, `apk add`, `adduser`, + the `chown`, every `go.mod`/`go.sum` and source copy, the linter copy out of + the lint stage, and the binary copy into the runtime stage. The lint stage + still gates the build stage: with a deliberate `unused` finding planted in the + tree, the build failed at `make lint` in 36.1s and the build-stage + `make check` never started. The build stage also still drops to the + unprivileged `builder` user before `make check`, which the suite depends on + rather than merely prefers: forcing the same image to run the tests as root + fails `TestScanHardlinkRunFailsTogether`, because root reads straight through + the `chmod(0)` the test uses to prove hard links are read once. This is the + local fix only; propagating it to the canonical templates is `prompts` #26 +- check the installed golangci-lint version in `script/bootstrap` instead of + only its presence (2026-08-09, branch `bootstrap-version-check`, closes #24): + `missing golangci-lint` meant any linter already on `PATH` satisfied the + check, so the pin was never consulted and the v2.12.2 bump from #3 was inert + on every host that already had one — this host ran v2.10.1 against a v2.12.2 + pin, `make check` went green, and `make docker` then rejected the same commit + with findings the local gate never saw. The version now lives in one place, + `GOLANGCI_LINT_VERSION`, with the `go install` module ref derived from it so a + bump cannot half-apply; a `golangci_lint_version` helper parses + `golangci-lint --version` (taking the field after the word `version` and + tolerating an optional leading `v`, which the module ref carries and the + binary's output does not), and any version that is not the pin — older, newer, + absent or unparseable — is reinstalled. The install is then verified against + the binary `PATH` actually resolves: `go install` writes into `GOBIN` (or + `GOPATH/bin`) while `make lint` runs whichever `golangci-lint` comes first on + `PATH`, so a wrong-version one sitting ahead of it — nix, apt, brew, apk, or + the `/usr/local/bin` copy the `Dockerfile` builder stage makes — would swallow + the install and leave the local gate disagreeing with CI under an affirmative + `bootstrap complete`. Bootstrap now re-reads the effective version after + installing and, on a mismatch, prints both paths and both versions to stderr + and exits non-zero instead of claiming success; it does not reorder anyone's `PATH` or delete their binary. The `--version` call keeps its stderr - connected, so a present-but-broken binary says why rather than - reinstalling forever in silence, and is bounded by `timeout(1)` where - that exists, so a wedged binary cannot hang bootstrap. `git`, `make` - and `go` keep their presence-only checks and now say why in a - comment: they are host package-manager tools the repo deliberately - does not pin, with `go.mod` governing the language version and the - digest-pinned images covering reproducible builds. Verified on this - host by bootstrapping from v2.10.1 to v2.12.2 and running it again to - a no-op, plus stub runs of the real script under `dash` covering a - thirteen-input parse matrix (absent, older, newer, host-style, - image-style, leading-`v`, stderr-only, empty, non-zero exit, impostor - binary, `(devel)`, trailing `version`), a shadowed install that must - exit non-zero, an install destination not on `PATH` at all, `GOBIN` - set, and a wedged binary that must hit the timeout; `make check` and - `make lint` are clean at v2.12.2, so v2.10.1 was not hiding any - findings on `main` + connected, so a present-but-broken binary says why rather than reinstalling + forever in silence, and is bounded by `timeout(1)` where that exists, so a + wedged binary cannot hang bootstrap. `git`, `make` and `go` keep their + presence-only checks and now say why in a comment: they are host + package-manager tools the repo deliberately does not pin, with `go.mod` + governing the language version and the digest-pinned images covering + reproducible builds. Verified on this host by bootstrapping from v2.10.1 to + v2.12.2 and running it again to a no-op, plus stub runs of the real script + under `dash` covering a thirteen-input parse matrix (absent, older, newer, + host-style, image-style, leading-`v`, stderr-only, empty, non-zero exit, + impostor binary, `(devel)`, trailing `version`), a shadowed install that must + exit non-zero, an install destination not on `PATH` at all, `GOBIN` set, and a + wedged binary that must hit the timeout; `make check` and `make lint` are + clean at v2.12.2, so v2.10.1 was not hiding any findings on `main` - unwind the hash worker pool on the error path (2026-08-09, branch - `hash-pool-cleanup`, closes #6): `hashPhase` used to return the - moment `recordRun` failed and abandon the pool — the feeder parked - forever on a full `jobs` channel and every worker on a full - `results` channel. That only stopped being invisible when #4 landed - and `runScan` began unwinding instead of calling `os.Exit`. The - pool is now an owned, context-aware `hashPool`: every blocking send - in the feeder and the workers selects on `ctx.Done()`, `jobs` is - closed on every path out, and `hashPhase` defers `pool.stop()`, - which cancels and then drains `results` until the last goroutine - has exited — draining is what frees a worker already parked on a - send. `ctx` is threaded from `cmd.Context()` through `runScan`, - `syncScan`, both worker pools and the whole database layer (it is - the first parameter everywhere), so #5 can hand this path a signal - and needs to add nothing else. The walk pool never leaked, because - `walkPhase` always drains its events to close, but it has the same - unbounded-send shape and #5 will give it an early return, so it - gets the same treatment plus a `ctx.Err()` guard after the walk: a - cancelled walk yields a partial size census, and every file it never - reached looks vanished to the update phase. That phase's own - `BeginTx` fails on the same cancelled context before deleting - anything, so the guard is defence in depth rather than the only - barrier — but it is the one that survives #5 deciding an interrupted - scan may commit what it has. Tests drive `run(scan)` against a - database whose insert trigger aborts, and assert both that the scan - fails instead of hanging and that `runtime.NumGoroutine()` polls - back to its pre-scan baseline; a second set cancels a scan part-way - through the walk — deterministically, by counting the scan's own - consultations of `ctx.Done()` rather than racing a timer — and - asserts that it stops at the guard holding a partial census and a - still-populated record index, with every record intact. The - remaining cancellation branches of both pools are covered by direct - tests of `sendEvent`, the walk workers, `dispatchDirs`, - `feedHashJobs`, `hashWorker` and `hashPhase` -- guarantee the database is closed on every fatal exit path - (2026-08-09, branch `db-close-on-fatal`, closes #4): `fatalf` and - its `os.Exit(1)` are gone, so the deferred `db.Close()` — and with - it the SQLite WAL checkpoint — now actually runs when a subcommand - fails; `runScan`, `runReport`, `runTrees`, `loadRecords` and - `resolveRoots` return errors instead. The single exit point is `run` - in `main.go`: it maps a `fatalError` (anything a subcommand - returned) to exit 1 and cobra's own argument and flag errors to exit - 2, which keeps a runtime failure from being reported as a usage - error or printing the usage text. New `main_test.go` drives the CLI - in-process and asserts the exit codes from README §Error handling - plus the stdout/stderr split, including that a fatal error raised - after the database is open leaves no `-wal`/`-shm` sidecar behind - for `scan`, `report` or `trees` -- update golangci-lint to v2.12.2 with the canonical config - (2026-08-09, branch `golangci-v2.12.2`, merged as `38a01bd`, - closes #3): bumped the pinned linter in the `Dockerfile` lint - stage and `script/bootstrap` from v2.12.1 to v2.12.2, and replaced - `.golangci.yml` with the canonical file — the linter settings + `hash-pool-cleanup`, closes #6): `hashPhase` used to return the moment + `recordRun` failed and abandon the pool — the feeder parked forever on a full + `jobs` channel and every worker on a full `results` channel. That only stopped + being invisible when #4 landed and `runScan` began unwinding instead of + calling `os.Exit`. The pool is now an owned, context-aware `hashPool`: every + blocking send in the feeder and the workers selects on `ctx.Done()`, `jobs` is + closed on every path out, and `hashPhase` defers `pool.stop()`, which cancels + and then drains `results` until the last goroutine has exited — draining is + what frees a worker already parked on a send. `ctx` is threaded from + `cmd.Context()` through `runScan`, `syncScan`, both worker pools and the whole + database layer (it is the first parameter everywhere), so #5 can hand this + path a signal and needs to add nothing else. The walk pool never leaked, + because `walkPhase` always drains its events to close, but it has the same + unbounded-send shape and #5 will give it an early return, so it gets the same + treatment plus a `ctx.Err()` guard after the walk: a cancelled walk yields a + partial size census, and every file it never reached looks vanished to the + update phase. That phase's own `BeginTx` fails on the same cancelled context + before deleting anything, so the guard is defence in depth rather than the + only barrier — but it is the one that survives #5 deciding an interrupted scan + may commit what it has. Tests drive `run(scan)` against a database whose + insert trigger aborts, and assert both that the scan fails instead of hanging + and that `runtime.NumGoroutine()` polls back to its pre-scan baseline; a + second set cancels a scan part-way through the walk — deterministically, by + counting the scan's own consultations of `ctx.Done()` rather than racing a + timer — and asserts that it stops at the guard holding a partial census and a + still-populated record index, with every record intact. The remaining + cancellation branches of both pools are covered by direct tests of + `sendEvent`, the walk workers, `dispatchDirs`, `feedHashJobs`, `hashWorker` + and `hashPhase` +- guarantee the database is closed on every fatal exit path (2026-08-09, branch + `db-close-on-fatal`, closes #4): `fatalf` and its `os.Exit(1)` are gone, so + the deferred `db.Close()` — and with it the SQLite WAL checkpoint — now + actually runs when a subcommand fails; `runScan`, `runReport`, `runTrees`, + `loadRecords` and `resolveRoots` return errors instead. The single exit point + is `run` in `main.go`: it maps a `fatalError` (anything a subcommand returned) + to exit 1 and cobra's own argument and flag errors to exit 2, which keeps a + runtime failure from being reported as a usage error or printing the usage + text. New `main_test.go` drives the CLI in-process and asserts the exit codes + from README §Error handling plus the stdout/stderr split, including that a + fatal error raised after the database is open leaves no `-wal`/`-shm` sidecar + behind for `scan`, `report` or `trees` +- update golangci-lint to v2.12.2 with the canonical config (2026-08-09, branch + `golangci-v2.12.2`, merged as `38a01bd`, closes #3): bumped the pinned linter + in the `Dockerfile` lint stage and `script/bootstrap` from v2.12.1 to v2.12.2, + and replaced `.golangci.yml` with the canonical file — the linter settings (`lll`, `funlen`, `cyclop`, `dupl` thresholds) now live under - `linters.settings` per the v2 schema, so they are actually - applied; no new lint findings surfaced -- convert Makefile targets to scripts-to-rule-them-all `script/` - entrypoints like the other managed repos (2026-07-26, commit - `3abeacf`, closes #1): all 12 `script/` entrypoints exist - (`bootstrap`, `setup`, `projectname`, `test`, `lint`, `fmt`, - `fmt-check`, `check`, `docker`, `cibuild`, `precommit`, - `install-precommit`) and every Makefile target is now a thin shim - over them, matching the other managed repos + `linters.settings` per the v2 schema, so they are actually applied; no new + lint findings surfaced +- convert Makefile targets to scripts-to-rule-them-all `script/` entrypoints + like the other managed repos (2026-07-26, commit `3abeacf`, closes #1): all 12 + `script/` entrypoints exist (`bootstrap`, `setup`, `projectname`, `test`, + `lint`, `fmt`, `fmt-check`, `check`, `docker`, `cibuild`, `precommit`, + `install-precommit`) and every Makefile target is now a thin shim over them, + matching the other managed repos - make the binary the default Make target (2026-07-24, branch - `make-default-target`): plain `make` now builds `sfdupes` - (previously it ran `check` plus `build`); `make build` remains as - an alias -- scan-wide phases, concurrent operands, batched updates (2026-07-24, - branch `scan-wide-phases`): all operands seed the shared walk pool - and every pass runs once over the whole scan, so totals and ETAs - are scan-global; the per-operand walk/hash/update cycles and their - stderr announcements are gone; the update pass commits in batched - transactions — the filesystem is authoritative and the database an - eventually-consistent reflection, so scan-level atomicity is not - required + `make-default-target`): plain `make` now builds `sfdupes` (previously it ran + `check` plus `build`); `make build` remains as an alias +- scan-wide phases, concurrent operands, batched updates (2026-07-24, branch + `scan-wide-phases`): all operands seed the shared walk pool and every pass + runs once over the whole scan, so totals and ETAs are scan-global; the + per-operand walk/hash/update cycles and their stderr announcements are gone; + the update pass commits in batched transactions — the filesystem is + authoritative and the database an eventually-consistent reflection, so + scan-level atomicity is not required - split the stat pass back out of the walk (2026-07-24, branch - `parallel-phases`): phases are strictly sequential again — walk, - stat, hash, update per operand — with parallelism only inside each - phase; the walk enumerates paths with per-directory workers and the - stat pass lstats them with per-file workers, restoring the exact - total/ETA stat bar -- announce each operand on stderr before its passes (2026-07-24, - branch `scan-operand-progress`): with per-operand walk/hash/update - cycles, a multi-operand run (e.g. `scan /srv/*`) showed pass totals - that looked like the whole run's — an operator watching operand 3 of - 14 hash 300k files concluded 20M files were being skipped -- parallel walk (2026-07-24, branch `parallel-walk`): the walk pass - was a single goroutine and took hours at ~20M files on a busy pool - (observed: 22M files in 4h on a ZFS server); it is now a - per-directory worker-pool traversal that records size/mtime during - the walk (folding away the separate stat pass, halving metadata - I/O), and each `PATH` operand commits in its own transaction so an - interrupted scan keeps completed operands + `parallel-phases`): phases are strictly sequential again — walk, stat, hash, + update per operand — with parallelism only inside each phase; the walk + enumerates paths with per-directory workers and the stat pass lstats them with + per-file workers, restoring the exact total/ETA stat bar +- announce each operand on stderr before its passes (2026-07-24, branch + `scan-operand-progress`): with per-operand walk/hash/update cycles, a + multi-operand run (e.g. `scan /srv/*`) showed pass totals that looked like the + whole run's — an operator watching operand 3 of 14 hash 300k files concluded + 20M files were being skipped +- parallel walk (2026-07-24, branch `parallel-walk`): the walk pass was a single + goroutine and took hours at ~20M files on a busy pool (observed: 22M files in + 4h on a ZFS server); it is now a per-directory worker-pool traversal that + records size/mtime during the walk (folding away the separate stat pass, + halving metadata I/O), and each `PATH` operand commits in its own transaction + so an interrupted scan keeps completed operands -- persistent scan database (2026-07-24, branch `persistent-database`): - `scan` now maintains a SQLite database (`modernc.org/sqlite`, pure - Go, cgo stays disabled) keyed by absolute path that survives between - runs — a rescan hashes only new or changed files (by mtime/size), - deletes records for files vanished from under the scanned operands, - and leaves records outside them untouched, so `scan` can be cronned - daily; `report` and `trees` read the database (no positional - arguments) instead of a scan stream. Database at - `/var/lib/sfdupes/db.sqlite`, overridable via `SFDUPES_DATABASE`; - WAL journaling plus a single-transaction update keep a report run - during a scan safe -- add the `origin` remote (`git@git.eeqj.de:sneak/sfdupes.git`), tag - `v0.0.1`, and push `main` plus tags (2026-07-23) +- persistent scan database (2026-07-24, branch `persistent-database`): `scan` + now maintains a SQLite database (`modernc.org/sqlite`, pure Go, cgo stays + disabled) keyed by absolute path that survives between runs — a rescan hashes + only new or changed files (by mtime/size), deletes records for files vanished + from under the scanned operands, and leaves records outside them untouched, so + `scan` can be cronned daily; `report` and `trees` read the database (no + positional arguments) instead of a scan stream. Database at + `/var/lib/sfdupes/db.sqlite`, overridable via `SFDUPES_DATABASE`; WAL + journaling plus a single-transaction update keep a report run during a scan + safe +- add the `origin` remote (`git@git.eeqj.de:sneak/sfdupes.git`), tag `v0.0.1`, + and push `main` plus tags (2026-07-23) - `scan` CLI rework (2026-07-23, branch `scan-required-paths`): required - `PATH...` operands via cobra flags replacing the `/srv` `-root` - default; new `-x`/`--one-file-system` flag (GNU convention) to stop - at filesystem boundaries, which are crossed by default + `PATH...` operands via cobra flags replacing the `/srv` `-root` default; new + `-x`/`--one-file-system` flag (GNU convention) to stop at filesystem + boundaries, which are crossed by default - bring the repo into full policy compliance (2026-07-23, branch `repo-policy-compliance`; checklist below) -- `git init` with README-only first commit; code baseline committed on - `main` (2026-07-22) +- `git init` with README-only first commit; code baseline committed on `main` + (2026-07-22) - implement `scan`, `report`, and `trees` subcommands (pre-git history) # Future Steps -- possible later features (explicitly out of scope per README): - full-content verification of candidates, removal-script helpers +- possible later features (explicitly out of scope per README): full-content + verification of candidates, removal-script helpers # Repo Policy Compliance -Audited 2026-07-22 against `REPO_POLICIES.md` (2026-07-06), the existing -repo checklist, and the Go styleguide. Code is already gofmt-clean, so no -standalone formatting commit is needed. +Audited 2026-07-22 against `REPO_POLICIES.md` (2026-07-06), the existing repo +checklist, and the Go styleguide. Code is already gofmt-clean, so no standalone +formatting commit is needed. -- [x] `.gitignore` missing — the compiled `sfdupes` binary and - `files.dat` sit untracked in the tree; needs OS/editor/Go - artifacts plus secrets patterns +- [x] `.gitignore` missing — the compiled `sfdupes` binary and `files.dat` sit + untracked in the tree; needs OS/editor/Go artifacts plus secrets patterns - [x] `.editorconfig` missing -- [x] `LICENSE` missing and README has no License section (MIT assumed - from house convention — user to confirm) +- [x] `LICENSE` missing and README has no License section (MIT assumed from + house convention — user to confirm) - [x] `REPO_POLICIES.md` missing from repo root -- [x] `.golangci.yml` missing (install canonical copy); code must then - pass `make lint` (150 findings fixed; `make lint` is clean) -- [x] `Makefile` lacks required targets `test`, `lint`, `fmt`, - `fmt-check`, `docker`, `hooks`; `check` currently depends on - `build`, which writes the binary (`make check` must not modify - files) -- [x] no tests — `go test ./...` has nothing to run; policy requires - real tests with a 30-second timeout and the conditional `-v` - rerun pattern (suite covers parsing, grouping, digests, - suppression, hashing, and the scan pipeline; 64% coverage) -- [x] `Dockerfile` missing — Go multistage with hash-pinned images: - fail-fast lint stage, build stage running `make check` +- [x] `.golangci.yml` missing (install canonical copy); code must then pass + `make lint` (150 findings fixed; `make lint` is clean) +- [x] `Makefile` lacks required targets `test`, `lint`, `fmt`, `fmt-check`, + `docker`, `hooks`; `check` currently depends on `build`, which writes the + binary (`make check` must not modify files) +- [x] no tests — `go test ./...` has nothing to run; policy requires real tests + with a 30-second timeout and the conditional `-v` rerun pattern (suite + covers parsing, grouping, digests, suppression, hashing, and the scan + pipeline; 64% coverage) +- [x] `Dockerfile` missing — Go multistage with hash-pinned images: fail-fast + lint stage, build stage running `make check` - [x] `.dockerignore` missing -- [x] `.gitea/workflows/check.yml` missing (`docker build .` on push, - checkout action pinned by commit SHA) +- [x] `.gitea/workflows/check.yml` missing (`docker build .` on push, checkout + action pinned by commit SHA) - [x] README lacks required sections: Description first line - (name/purpose/category/license/author), Getting Started, - Rationale, TODO, License, Author -- [x] README non-goal "no git repository setup and no CI" is stale now - that the repo is under git with CI -- [x] pre-commit hook not installed (`make hooks` once the target - exists) + (name/purpose/category/license/author), Getting Started, Rationale, TODO, + License, Author +- [x] README non-goal "no git repository setup and no CI" is stale now that the + repo is under git with CI +- [x] pre-commit hook not installed (`make hooks` once the target exists) Accepted divergences (no action): -- flat single-package layout with `.go` files in the repo root — fine - for a small single-binary tool per the Go styleguide; the tracker - audit agrees -- `go test` runs without `-race` — the repo mandates `CGO_ENABLED=0` - (pure-Go builds) and the race detector requires cgo +- flat single-package layout with `.go` files in the repo root — fine for a + small single-binary tool per the Go styleguide; the tracker audit agrees +- `go test` runs without `-race` — the repo mandates `CGO_ENABLED=0` (pure-Go + builds) and the race detector requires cgo diff --git a/package.json b/package.json new file mode 100644 index 0000000..dc05cde --- /dev/null +++ b/package.json @@ -0,0 +1,5 @@ +{ + "devDependencies": { + "prettier": "3.8.1" + } +} diff --git a/script/bootstrap b/script/bootstrap index 936ea8b..380191d 100755 --- a/script/bootstrap +++ b/script/bootstrap @@ -3,10 +3,12 @@ # this repo. Idempotent: every install is guarded by a check so already # installed tools are skipped. Base tooling comes from nix, apt, brew, # or apk (detected in that order); assumes nothing is present (not git, -# make, or go). The linter is NOT installed: golangci-lint runs via -# docker only (script/lint), pinned by image digest, so the only lint +# make, or go). Neither the linter nor the Markdown formatter is +# installed: golangci-lint (script/lint) and prettier (script/fmt, +# script/fmt-check) run via docker only, pinned by hash, so their only # prerequisite is a working docker — which is warned about, not -# installed, because everything except linting works without it. +# installed, because everything except linting and formatting works +# without it. set -eu ROOT="$(cd "$(dirname "$0")/.." && pwd -P)" @@ -71,15 +73,15 @@ main() { if missing make; then pkg_install gnumake make make make; fi if missing go; then pkg_install go golang go go; fi - # Linting runs via docker only (script/lint), so docker is a lint - # prerequisite rather than something bootstrap installs. Warn, do - # not fail: everything except `make lint` — and, through it, - # `make check`, `make docker` and the pre-commit hook — works - # without it. + # Linting and Markdown formatting run via docker only, so docker is + # their prerequisite rather than something bootstrap installs. Warn, + # do not fail: everything except `make lint`, `make fmt` and + # `make fmt-check` — and, through them, `make check`, `make docker` + # and the pre-commit hook — works without it. if missing docker; then - echo "bootstrap: WARNING: docker not found; make lint, make check" >&2 - echo "bootstrap: and make docker require it. Install docker to" >&2 - echo "bootstrap: run the linter." >&2 + echo "bootstrap: WARNING: docker not found; make lint, make fmt," >&2 + echo "bootstrap: make fmt-check, make check and make docker" >&2 + echo "bootstrap: require it." >&2 fi go mod download diff --git a/script/cibuild b/script/cibuild index 55a90ba..7283a88 100755 --- a/script/cibuild +++ b/script/cibuild @@ -3,17 +3,17 @@ # push. # # The Dockerfile runs the gates individually as build steps, not the -# make check aggregate: the lint stage runs make fmt-check, +# make check aggregate: the lint stage runs the gofmt check, # script/verify-lint-image-pin, golangci-lint config verify and -# golangci-lint run; the build stage, dropped to an unprivileged user, -# runs make test and make fmt-check. Neither make lint nor make check -# appears, because both reach script/lint, which is itself a docker -# build, and a docker build cannot run inside one. Lint is not skipped -# by that — the linter is invoked directly in the lint stage, and the -# build stage's COPY --from=lint makes that stage a prerequisite, so -# BuildKit must finish it first. Between the two stages everything -# make check would run has run, which is why a successful build here -# implies the repo is green. +# golangci-lint run; the markdown stage runs the prettier check; the +# build stage, dropped to an unprivileged user, runs make test. None of +# make lint, make fmt-check or make check appears, because each runs +# docker, and docker cannot run inside a docker build. Nothing is +# skipped by that — the linter, gofmt and prettier are invoked directly +# in their stages, and the build stage's COPY --from lines make those +# stages prerequisites, so BuildKit must finish them first. Between the +# three stages everything make check would run has run, which is why a +# successful build here implies the repo is green. # # That implication holds only because of CHECK_EPOCH. A COPY layer is # invalidated only by changed content, and a rebuild of an unchanged diff --git a/script/docker b/script/docker index 9f52294..3c7f7bc 100755 --- a/script/docker +++ b/script/docker @@ -4,11 +4,11 @@ # # CHECK_EPOCH is passed for the same reason script/cibuild passes it: # without it Docker serves the Dockerfile's gate layers from cache on an -# unchanged tree and this exits 0 having run neither the lint stage's -# gates nor the builder stage's test and fmt-check gates. This is the -# set of gates a developer or reviewer runs by hand, so a cached pass -# here is the most misleading result the repo can produce. Dependency -# layers sit above the ARG and stay cached. +# unchanged tree and this exits 0 having run none of the lint stage's +# gates, the markdown stage's prettier gate or the builder stage's test +# gate. This is the set of gates a developer or reviewer runs by hand, +# so a cached pass here is the most misleading result the repo can +# produce. Dependency layers sit above the ARG and stay cached. set -eu SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)" diff --git a/script/fmt b/script/fmt index 28af13b..8994c8b 100755 --- a/script/fmt +++ b/script/fmt @@ -1,12 +1,22 @@ #!/bin/sh -# script/fmt: format all files (writes). +# script/fmt: format all files (writes): the Go sources with gofmt, the +# Markdown with prettier. prettier is never installed on the host: it +# runs from the Dockerfile's prettier stage with the repository mounted, +# as the calling user so the files it rewrites keep their owner. The tag +# makes each build replace the previous image instead of leaving another +# one behind. set -eu -ROOT="$(cd "$(dirname "$0")/.." && pwd -P)" +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)" +ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)" main() { cd "$ROOT" gofmt -s -w . + image="$("$SCRIPT_DIR/projectname")-prettier" + docker build -q --target prettier -t "$image" . >/dev/null + docker run --rm --user "$(id -u):$(id -g)" -v "$ROOT:/src" "$image" \ + prettier --write '**/*.md' --tab-width 4 --prose-wrap always } main "$@" diff --git a/script/fmt-check b/script/fmt-check index 96fffa0..8cf37b3 100755 --- a/script/fmt-check +++ b/script/fmt-check @@ -1,18 +1,34 @@ #!/bin/sh # script/fmt-check: check formatting (read-only). Same scope as -# script/fmt, but fails instead of writing. +# script/fmt, but fails instead of writing. gofmt and prettier both run +# every time and each reports its own failure, so the output says which +# one failed. set -eu -ROOT="$(cd "$(dirname "$0")/.." && pwd -P)" +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)" +ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)" main() { cd "$ROOT" + status=0 + files="$(gofmt -s -l .)" if [ -n "$files" ]; then echo "gofmt: files not formatted:" >&2 echo "$files" >&2 - exit 1 + status=1 fi + + # Same image as script/fmt; see there. + image="$("$SCRIPT_DIR/projectname")-prettier" + docker build -q --target prettier -t "$image" . >/dev/null + if ! docker run --rm -v "$ROOT:/src:ro" "$image" \ + prettier --check '**/*.md' --tab-width 4 --prose-wrap always; then + echo "prettier: Markdown not formatted; run make fmt" >&2 + status=1 + fi + + exit "$status" } main "$@" diff --git a/yarn.lock b/yarn.lock new file mode 100644 index 0000000..d846639 --- /dev/null +++ b/yarn.lock @@ -0,0 +1,8 @@ +# THIS IS AN AUTOGENERATED FILE. DO NOT EDIT THIS FILE DIRECTLY. +# yarn lockfile v1 + + +prettier@3.8.1: + version "3.8.1" + resolved "https://registry.yarnpkg.com/prettier/-/prettier-3.8.1.tgz#edf48977cf991558f4fcbd8a3ba6015ba2a3a173" + integrity sha512-UOnG6LftzbdaHZcKoPFtOcCKztrQ57WkHDeRD9t/PTQtmT0NHSeWWepj6pS0z/N7+08BHFDQVUrfmfMRcZwbMg==