Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2acb657a44 | ||
|
|
5900feb515 |
@@ -13,3 +13,6 @@ indent_style = tab
|
||||
|
||||
[*.go]
|
||||
indent_style = tab
|
||||
|
||||
# This repository's own sections, such as one for another language it
|
||||
# uses, go below this comment, and a re-vendor keeps them.
|
||||
|
||||
@@ -1,11 +1,20 @@
|
||||
name: check
|
||||
on: [push]
|
||||
# Free the shared runner: a new push cancels only the same branch's older run.
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
jobs:
|
||||
check:
|
||||
runs-on: ubuntu-latest
|
||||
# Free the shared runner from a hung build.
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
# actions/checkout v4.2.2, 2026-02-22
|
||||
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683
|
||||
# script/cibuild needs no token, so none is left in .git/config.
|
||||
with:
|
||||
persist-credentials: false
|
||||
# All history and tags, so git describe finds the version tag.
|
||||
fetch-depth: 0
|
||||
- run: script/cibuild
|
||||
|
||||
+5
-3
@@ -27,7 +27,7 @@ node_modules/
|
||||
# Environment files. `*.env` covers bare `.env` and the `prod.env`
|
||||
# convention. Only the templates `example.env` and `sample.env` are
|
||||
# re-included below. A repository that commits any other template adds
|
||||
# its own negation after these lines, for example `!.env.example`.
|
||||
# its own negation at the end of this file, for example `!.env.example`.
|
||||
*.[eE][nN][vV]
|
||||
.[eE][nN][vV].*
|
||||
.[eE][nN][vV][rR][cC]
|
||||
@@ -46,11 +46,13 @@ node_modules/
|
||||
[iI][dD]_[eE][dD]25519
|
||||
[iI][dD]_[eE][dD]25519_[sS][kK]
|
||||
|
||||
# This repository's own entries, kept after the canonical content above.
|
||||
# This repository's own entries, such as its build outputs, go below
|
||||
# this comment, and a re-vendor keeps them. Anchor a binary built at the
|
||||
# root: `/myapp`, never `myapp`, which also ignores `cmd/myapp/`.
|
||||
/sfdupes
|
||||
*.log
|
||||
*.out
|
||||
*.test
|
||||
/sfdupes
|
||||
|
||||
# A scan database lists every path it scanned.
|
||||
*.sqlite
|
||||
|
||||
@@ -25,6 +25,8 @@ linters:
|
||||
# silenced by disabling that name, not by enabling the successor.
|
||||
- wsl # Deprecated, replaced by wsl_v5
|
||||
- gomodguard # Deprecated, replaced by gomodguard_v2
|
||||
# Misses findings at random in v2.14.0; back once a pinned release fixes it
|
||||
- canonicalheader
|
||||
settings:
|
||||
lll:
|
||||
line-length: 88
|
||||
|
||||
@@ -312,29 +312,34 @@ All three subcommands operate on a single SQLite database file:
|
||||
|
||||
```sql
|
||||
CREATE TABLE files (
|
||||
path BLOB PRIMARY KEY, -- absolute path, raw bytes
|
||||
size INTEGER NOT NULL, -- bytes, from lstat
|
||||
mtime INTEGER NOT NULL, -- Unix seconds, from lstat
|
||||
head TEXT NOT NULL, -- lowercase-hex SHA-256; first 64 KiB, or whole file under 10 MiB
|
||||
tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB
|
||||
content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples
|
||||
path BLOB PRIMARY KEY, -- absolute path, raw bytes
|
||||
size INTEGER NOT NULL, -- bytes, from lstat
|
||||
mtime INTEGER NOT NULL, -- whole Unix seconds of the mtime, from lstat
|
||||
mtime_nsec INTEGER NOT NULL, -- nanoseconds within that second, 0 to 999999999
|
||||
head TEXT NOT NULL, -- lowercase-hex SHA-256; first 64 KiB, or whole file under 10 MiB
|
||||
tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB
|
||||
content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples
|
||||
) WITHOUT ROWID;
|
||||
CREATE INDEX files_signature ON files (size, head, tail, content);
|
||||
```
|
||||
|
||||
Paths are stored as BLOBs because Unix paths are raw bytes, not guaranteed
|
||||
UTF-8. `mtime` is used only for change detection; it is not part of the
|
||||
duplicate key. For a file under 10 MiB `head`, `tail`, and `content` all
|
||||
hold the whole-file hash (that range is hashed in full, with no end
|
||||
windows); for a larger file `head` and `tail` hold the first- and last-64
|
||||
KiB hashes and `content` the whole-file or sampled hash. All three are empty
|
||||
strings when the file has never been hashed because its size was unique as
|
||||
of the last scan that covered it. For a file of 10 MiB or more, `content`
|
||||
stays empty until the content phase of a scan (see "`scan` mode" below) has
|
||||
read the file. A record with an empty `content` is never part of a duplicate
|
||||
group, though it still defines the file for tree reconstruction. The
|
||||
`files_signature` index lets SQLite group the records by signature for
|
||||
`report` without sorting the whole table.
|
||||
UTF-8. `mtime` and `mtime_nsec` hold the file's mtime to the nanosecond:
|
||||
`mtime` the whole Unix seconds, rounded down, and `mtime_nsec` the
|
||||
nanoseconds past that second. Split this way they hold any mtime a
|
||||
filesystem can record, one before 1678 or after 2262 included, which a
|
||||
single 64-bit count of nanoseconds cannot. They are used only for change
|
||||
detection and are not part of the duplicate key. For a file under 10 MiB
|
||||
`head`, `tail`, and `content` all hold the whole-file hash (that range is
|
||||
hashed in full, with no end windows); for a larger file `head` and `tail`
|
||||
hold the first- and last-64 KiB hashes and `content` the whole-file or
|
||||
sampled hash. All three are empty strings when the file has never been
|
||||
hashed because its size was unique as of the last scan that covered it. For
|
||||
a file of 10 MiB or more, `content` stays empty until the content phase of a
|
||||
scan (see "`scan` mode" below) has read the file. A record with an empty
|
||||
`content` is never part of a duplicate group, though it still defines the
|
||||
file for tree reconstruction. The `files_signature` index lets SQLite group
|
||||
the records by signature for `report` without sorting the whole table.
|
||||
|
||||
### Duplicate detection
|
||||
|
||||
@@ -421,6 +426,9 @@ operands:
|
||||
once its size, `head`, and `tail` match another record's.
|
||||
- A file whose mtime is newer than recorded, or whose size differs, is processed
|
||||
as if new: re-hashed, or recorded without hashes, per the shared-size rule.
|
||||
Change detection compares the mtime to the nanosecond, as finely as the
|
||||
filesystem records it, so a same-size rewrite counts as a change whenever the
|
||||
filesystem gives it a later mtime than recorded, even within the same second.
|
||||
- A database record whose path lies under one of the scanned operands but was
|
||||
not successfully processed this run is deleted. This removes records for
|
||||
deleted files. It also removes records for paths that failed to stat or hash
|
||||
@@ -740,11 +748,12 @@ entrypoints are:
|
||||
come from the first of nix, apt, brew, or apk found on the host, and are
|
||||
presence-checked only. An installed node is used as it is; otherwise node
|
||||
22.17.0 is installed through nvm, which comes from a release archive whose
|
||||
sha256 the script checks. yarn 1.22.22 comes through corepack, and
|
||||
`yarn install --frozen-lockfile` installs the prettier that `package.json` and
|
||||
`yarn.lock` pin. `golangci-lint` is never installed: it runs in Docker (see
|
||||
`script/lint`). `docker` is not installed either; testing, linting and the
|
||||
image build need it. Ends with `go mod download`.
|
||||
sha256 the script checks. A yarn already on `PATH` is used as it is; otherwise
|
||||
yarn 1.22.22 is activated through corepack, or installed with `npm` when there
|
||||
is no corepack. `yarn install --frozen-lockfile` then installs the prettier
|
||||
that `package.json` and `yarn.lock` pin. `golangci-lint` is never installed:
|
||||
it runs in Docker (see `script/lint`). `docker` is not installed either;
|
||||
testing, linting and the image build need it. Ends with `go mod download`.
|
||||
- `script/setup` — make a fresh clone ready for development: runs
|
||||
`script/bootstrap`, then `script/install-precommit`.
|
||||
- `script/projectname` — print this project's name (`sfdupes`). Scripts that
|
||||
@@ -786,9 +795,8 @@ entrypoints are:
|
||||
- `script/precommit` — run by the git pre-commit hook: `go mod tidy` must be a
|
||||
no-op (a resulting change to `go.mod` or `go.sum` fails the commit), then
|
||||
`script/check`.
|
||||
- `script/install-precommit` — install the git pre-commit hook that runs
|
||||
`script/precommit`. The hook is written to the common git directory, so the
|
||||
main checkout and every worktree share it.
|
||||
- `script/install-precommit` — install the git pre-commit hook, as
|
||||
`.git/hooks/pre-commit`, that runs `script/precommit`.
|
||||
|
||||
Every `docker build` in `script/` passes `--no-cache`. On an unchanged tree
|
||||
Docker would serve the gate steps from its cache, and the build would pass
|
||||
|
||||
@@ -29,24 +29,32 @@
|
||||
|
||||
# Completed Steps
|
||||
|
||||
- re-vendor the canonical files from `sneak/prompts` commit `dd4027b`, with
|
||||
`script/lint`, `script/test` and `REPO_POLICIES.md` from its `next` at
|
||||
- re-vendor the canonical files and model scripts from `sneak/prompts` `next` at
|
||||
`c55a0cb`: golangci-lint v2.14.0; lint and test are phases of the `Dockerfile`
|
||||
that write no image, and `make test` runs the suite under the race detector,
|
||||
so `Dockerfile.lint`, `script/verify-lint-image-pin` and `make test-race` are
|
||||
gone; every `docker build` in `script/` passes `--no-cache`; prettier runs on
|
||||
the host, from the node and yarn `script/bootstrap` installs, so the
|
||||
`prettier` and `markdown` stages are gone and the build stage installs `git`
|
||||
and `make` itself; `.claude/settings.json` is deleted (2026-10-08,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/95)
|
||||
and `make` itself; a new push cancels the workflow's older run on the same
|
||||
branch, and a run stops after 20 minutes; `.claude/settings.json` is deleted
|
||||
(2026-10-08, https://git.eeqj.de/sneak/sfdupes/issues/95)
|
||||
|
||||
- `scan` records mtime to the nanosecond, as whole seconds in `mtime` plus
|
||||
`mtime_nsec`, and compares it at that resolution, so a same-size rewrite
|
||||
within the same second is re-hashed (2026-10-07,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/12)
|
||||
|
||||
- cut the narration from `TODO.md` Completed Steps and from the comments in
|
||||
`script/` and both Dockerfiles; §Workflow now branches from and merges to
|
||||
`next` (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/49)
|
||||
|
||||
- `make test-race` runs the test suite under the race detector in a cgo-enabled
|
||||
- `make test-race` ran the test suite under the race detector in a cgo-enabled
|
||||
container, outside `make check` (2026-10-04,
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/18)
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/18). Since
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/95 `make test` itself runs the suite
|
||||
under the race detector, in the `Dockerfile`'s Debian-based `test` phase, and
|
||||
`make test-race` is gone
|
||||
|
||||
- a bare `docker build .` failed, naming `script/cibuild` and `script/docker`,
|
||||
rather than serve the gates from cache (2026-10-04,
|
||||
@@ -99,9 +107,12 @@
|
||||
- README documents install, Docker, a daily cron scan and how to read and check
|
||||
the reports (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/54)
|
||||
|
||||
- the `Dockerfile` build stage keeps the Go module cache out of `builder`'s home
|
||||
and copies the sources with `--chown`, so no `chown -R` walks them
|
||||
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/43)
|
||||
- the `Dockerfile` build stage kept the Go module cache out of `builder`'s home
|
||||
and copied the sources with `--chown`, so no `chown -R` walked them
|
||||
(2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/43). Since
|
||||
https://git.eeqj.de/sneak/sfdupes/issues/95 there is no `builder` user and
|
||||
nothing changes owner: the build stage only compiles, as root, and the tests
|
||||
run as `nobody` in the `test` phase
|
||||
|
||||
- `--version` prints `sfdupes VERSION` to stdout; README documents it and
|
||||
`--help` (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/15)
|
||||
|
||||
@@ -11,6 +11,7 @@ import (
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"golang.org/x/sys/unix"
|
||||
// The pure-Go SQLite driver, registered as "sqlite"; keeps cgo
|
||||
@@ -40,15 +41,18 @@ const dbDirPerm = 0o755
|
||||
const lockFilePerm = 0o600
|
||||
|
||||
// createTableSQL is the schema applied to a fresh database. Paths are
|
||||
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8.
|
||||
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8. mtime
|
||||
// holds whole Unix seconds and mtime_nsec the nanoseconds within that
|
||||
// second.
|
||||
const createTableSQL = `
|
||||
CREATE TABLE files (
|
||||
path BLOB PRIMARY KEY,
|
||||
size INTEGER NOT NULL,
|
||||
mtime INTEGER NOT NULL,
|
||||
head TEXT NOT NULL,
|
||||
tail TEXT NOT NULL,
|
||||
content TEXT NOT NULL
|
||||
path BLOB PRIMARY KEY,
|
||||
size INTEGER NOT NULL,
|
||||
mtime INTEGER NOT NULL,
|
||||
mtime_nsec INTEGER NOT NULL,
|
||||
head TEXT NOT NULL,
|
||||
tail TEXT NOT NULL,
|
||||
content TEXT NOT NULL
|
||||
) WITHOUT ROWID
|
||||
`
|
||||
|
||||
@@ -61,10 +65,11 @@ CREATE INDEX files_signature ON files (size, head, tail, content)
|
||||
// upsertSQL inserts one file record, replacing any existing record for
|
||||
// the same path.
|
||||
const upsertSQL = `
|
||||
INSERT INTO files (path, size, mtime, head, tail, content)
|
||||
VALUES (?, ?, ?, ?, ?, ?)
|
||||
INSERT INTO files (path, size, mtime, mtime_nsec, head, tail, content)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT (path) DO UPDATE SET
|
||||
size = excluded.size, mtime = excluded.mtime,
|
||||
mtime_nsec = excluded.mtime_nsec,
|
||||
head = excluded.head, tail = excluded.tail,
|
||||
content = excluded.content
|
||||
`
|
||||
@@ -356,8 +361,8 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) {
|
||||
// which is the order of the primary key, so SQLite does not sort.
|
||||
func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
|
||||
rows, err := db.QueryContext(ctx,
|
||||
"SELECT path, size, mtime, head, tail, content FROM files "+
|
||||
"ORDER BY path")
|
||||
"SELECT path, size, mtime, mtime_nsec, head, tail, content "+
|
||||
"FROM files ORDER BY path")
|
||||
if err != nil {
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
@@ -366,17 +371,19 @@ func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
|
||||
|
||||
for rows.Next() {
|
||||
var (
|
||||
path []byte
|
||||
r scanRec
|
||||
path []byte
|
||||
sec, nsec int64
|
||||
r scanRec
|
||||
)
|
||||
|
||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
|
||||
err = rows.Scan(&path, &r.size, &sec, &nsec, &r.head, &r.tail,
|
||||
&r.content)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read record: %w", err)
|
||||
}
|
||||
|
||||
r.path = string(path)
|
||||
r.mtime = time.Unix(sec, nsec)
|
||||
fn(r)
|
||||
}
|
||||
|
||||
@@ -466,10 +473,10 @@ func loadDupeRows(ctx context.Context, db *sql.DB,
|
||||
// values, and skipping the hash columns keeps the scan's in-memory
|
||||
// index small on multi-million-file databases.
|
||||
func loadFileMeta(ctx context.Context, db *sql.DB,
|
||||
fn func(path string, size, mtime int64, hashed bool),
|
||||
fn func(path string, size int64, mtime time.Time, hashed bool),
|
||||
) error {
|
||||
rows, err := db.QueryContext(ctx,
|
||||
"SELECT path, size, mtime, head <> '' FROM files")
|
||||
"SELECT path, size, mtime, mtime_nsec, head <> '' FROM files")
|
||||
if err != nil {
|
||||
return fmt.Errorf("read records: %w", err)
|
||||
}
|
||||
@@ -478,17 +485,17 @@ func loadFileMeta(ctx context.Context, db *sql.DB,
|
||||
|
||||
for rows.Next() {
|
||||
var (
|
||||
path []byte
|
||||
size, mtime int64
|
||||
hashed int64
|
||||
path []byte
|
||||
size, sec, nsec int64
|
||||
hashed int64
|
||||
)
|
||||
|
||||
err = rows.Scan(&path, &size, &mtime, &hashed)
|
||||
err = rows.Scan(&path, &size, &sec, &nsec, &hashed)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read record: %w", err)
|
||||
}
|
||||
|
||||
fn(string(path), size, mtime, hashed != 0)
|
||||
fn(string(path), size, time.Unix(sec, nsec), hashed != 0)
|
||||
}
|
||||
|
||||
err = rows.Err()
|
||||
@@ -507,7 +514,7 @@ func loadFileMeta(ctx context.Context, db *sql.DB,
|
||||
// memory; the rows come ordered by size, head, and tail, so each
|
||||
// group's rows arrive together.
|
||||
const contentCandidatesSQL = `
|
||||
SELECT f.path, f.size, f.mtime, f.head, f.tail, f.content <> ''
|
||||
SELECT f.path, f.size, f.mtime, f.mtime_nsec, f.head, f.tail, f.content <> ''
|
||||
FROM files AS f
|
||||
JOIN (
|
||||
SELECT size, head, tail
|
||||
@@ -533,17 +540,20 @@ func loadContentCandidates(ctx context.Context, db *sql.DB,
|
||||
|
||||
for rows.Next() {
|
||||
var (
|
||||
path []byte
|
||||
r scanRec
|
||||
hashed int64
|
||||
path []byte
|
||||
sec, nsec int64
|
||||
r scanRec
|
||||
hashed int64
|
||||
)
|
||||
|
||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, &hashed)
|
||||
err = rows.Scan(&path, &r.size, &sec, &nsec, &r.head, &r.tail,
|
||||
&hashed)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read record: %w", err)
|
||||
}
|
||||
|
||||
r.path = string(path)
|
||||
r.mtime = time.Unix(sec, nsec)
|
||||
fn(r, hashed != 0)
|
||||
}
|
||||
|
||||
@@ -627,8 +637,8 @@ func execUpserts(ctx context.Context, tx *sql.Tx, upserts []scanRec,
|
||||
defer func() { _ = st.Close() }()
|
||||
|
||||
for _, r := range upserts {
|
||||
_, err = st.ExecContext(ctx,
|
||||
[]byte(r.path), r.size, r.mtime, r.head, r.tail, r.content)
|
||||
_, err = st.ExecContext(ctx, []byte(r.path), r.size,
|
||||
r.mtime.Unix(), r.mtime.Nanosecond(), r.head, r.tail, r.content)
|
||||
if err != nil {
|
||||
return fmt.Errorf("upsert %s: %w", r.path, err)
|
||||
}
|
||||
|
||||
+10
-5
@@ -10,6 +10,7 @@ import (
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// testDBPath returns a database path inside a fresh temp dir.
|
||||
@@ -260,10 +261,13 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
||||
// written.
|
||||
recs := []scanRec{
|
||||
{
|
||||
size: 2, mtime: 20, head: "h2", tail: "t2", content: "c2",
|
||||
path: "/a/tab\tnew\nline",
|
||||
size: 2, mtime: time.Unix(20, 999_999_999), head: "h2", tail: "t2",
|
||||
content: "c2", path: "/a/tab\tnew\nline",
|
||||
},
|
||||
{
|
||||
size: 1, mtime: time.Unix(10, 0), head: "h1", tail: "t1",
|
||||
content: "c1", path: "/a/x",
|
||||
},
|
||||
{size: 1, mtime: 10, head: "h1", tail: "t1", content: "c1", path: "/a/x"},
|
||||
}
|
||||
|
||||
err := applyChanges(t.Context(), db, recs, nil,
|
||||
@@ -281,7 +285,8 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
||||
// An upsert for an existing path updates in place; a delete
|
||||
// removes exactly its path.
|
||||
upd := scanRec{
|
||||
size: 3, mtime: 30, head: "h3", tail: "t3", content: "c3", path: "/a/x",
|
||||
size: 3, mtime: time.Unix(30, 0), head: "h3", tail: "t3", content: "c3",
|
||||
path: "/a/x",
|
||||
}
|
||||
|
||||
err = applyChanges(t.Context(), db, []scanRec{upd},
|
||||
@@ -308,7 +313,7 @@ func TestApplyChangesBatching(t *testing.T) {
|
||||
recs := make([]scanRec, 0, n)
|
||||
for i := range n {
|
||||
recs = append(recs, scanRec{
|
||||
size: int64(i), mtime: 1, head: "h", tail: "t",
|
||||
size: int64(i), mtime: time.Unix(1, 0), head: "h", tail: "t",
|
||||
path: fmt.Sprintf("/batch/%07d", i),
|
||||
})
|
||||
}
|
||||
|
||||
@@ -7,6 +7,7 @@ import (
|
||||
"io"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// ioBufSize is the buffer size for the buffered stdout writers.
|
||||
@@ -21,7 +22,7 @@ const minGroupSize = 2
|
||||
// only and used by scan for change detection.
|
||||
type scanRec struct {
|
||||
size int64
|
||||
mtime int64
|
||||
mtime time.Time
|
||||
head string
|
||||
tail string
|
||||
content string
|
||||
|
||||
+9
-2
@@ -11,6 +11,7 @@ import (
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// awkwardDir is a directory name holding every byte the reports escape.
|
||||
@@ -313,8 +314,14 @@ func TestDupeGroupsMtimeExcluded(t *testing.T) {
|
||||
// mtime is informational only; records differing only in mtime
|
||||
// still group together.
|
||||
recs := []scanRec{
|
||||
{size: 9, mtime: 100, head: "h", tail: "t", content: "c", path: "/m/1"},
|
||||
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
||||
{
|
||||
size: 9, mtime: time.Unix(100, 0), head: "h", tail: "t",
|
||||
content: "c", path: "/m/1",
|
||||
},
|
||||
{
|
||||
size: 9, mtime: time.Unix(200, 0), head: "h", tail: "t",
|
||||
content: "c", path: "/m/2",
|
||||
},
|
||||
}
|
||||
|
||||
groups := dupeGroupsOf(t, recs)
|
||||
|
||||
@@ -17,6 +17,7 @@ import (
|
||||
"strings"
|
||||
"sync"
|
||||
"syscall"
|
||||
"time"
|
||||
)
|
||||
|
||||
// The duplicate ladder (see hashSignature and README "Duplicate
|
||||
@@ -68,7 +69,7 @@ var errInterrupted = errors.New("scan interrupted")
|
||||
type fileRec struct {
|
||||
path string
|
||||
size int64
|
||||
mtime int64
|
||||
mtime time.Time
|
||||
dev uint64
|
||||
ino uint64
|
||||
}
|
||||
@@ -79,7 +80,7 @@ type fileRec struct {
|
||||
// they would dominate the scan's memory.
|
||||
type fileMeta struct {
|
||||
size int64
|
||||
mtime int64
|
||||
mtime time.Time
|
||||
hashed bool
|
||||
}
|
||||
|
||||
@@ -369,7 +370,7 @@ func (s *scanState) loadIndex(ctx context.Context, roots []string) error {
|
||||
s.existing = make(map[string]fileMeta)
|
||||
|
||||
return loadFileMeta(ctx, s.db,
|
||||
func(path string, size, mtime int64, hashed bool) {
|
||||
func(path string, size int64, mtime time.Time, hashed bool) {
|
||||
prog.increment()
|
||||
|
||||
if underAnyRoot(path, roots) {
|
||||
@@ -415,7 +416,7 @@ func (s *scanState) walkPhase(
|
||||
old, ok := s.existing[ev.rec.path]
|
||||
|
||||
switch {
|
||||
case !ok || old.size != ev.rec.size || old.mtime < ev.rec.mtime:
|
||||
case !ok || old.size != ev.rec.size || mtimeAfter(ev.rec.mtime, old.mtime):
|
||||
changed = append(changed, ev.rec)
|
||||
case old.hashed:
|
||||
delete(s.existing, ev.rec.path)
|
||||
@@ -807,7 +808,7 @@ func unchangedFile(r scanRec) (fileRec, bool, error) {
|
||||
}
|
||||
|
||||
if !fi.Mode().IsRegular() || fi.Size() != r.size ||
|
||||
fi.ModTime().Unix() > r.mtime {
|
||||
mtimeAfter(fi.ModTime(), r.mtime) {
|
||||
return fileRec{}, false, nil
|
||||
}
|
||||
|
||||
@@ -818,6 +819,16 @@ func unchangedFile(r scanRec) (fileRec, bool, error) {
|
||||
}, true, nil
|
||||
}
|
||||
|
||||
// mtimeAfter reports whether mtime a is later than mtime b.
|
||||
// Not a.After(b): time.Time wraps an mtime past year 292 billion; Unix() undoes it.
|
||||
func mtimeAfter(a, b time.Time) bool {
|
||||
if a.Unix() != b.Unix() {
|
||||
return a.Unix() > b.Unix()
|
||||
}
|
||||
|
||||
return a.Nanosecond() > b.Nanosecond()
|
||||
}
|
||||
|
||||
// underAnyRoot reports whether path is any of the roots or lies under
|
||||
// one of them.
|
||||
func underAnyRoot(path string, roots []string) bool {
|
||||
@@ -966,7 +977,7 @@ func seedRoot(ctx context.Context, root string,
|
||||
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
||||
path: root,
|
||||
size: fi.Size(),
|
||||
mtime: fi.ModTime().Unix(),
|
||||
mtime: fi.ModTime(),
|
||||
dev: dev,
|
||||
ino: ino,
|
||||
}})
|
||||
@@ -1124,7 +1135,7 @@ func emitFile(ctx context.Context, p string, e fs.DirEntry,
|
||||
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
||||
path: p,
|
||||
size: info.Size(),
|
||||
mtime: info.ModTime().Unix(),
|
||||
mtime: info.ModTime(),
|
||||
dev: dev,
|
||||
ino: ino,
|
||||
}})
|
||||
|
||||
+227
-7
@@ -20,6 +20,8 @@ import (
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"golang.org/x/sys/unix"
|
||||
)
|
||||
|
||||
// writeFile creates a file with the given content and returns its path.
|
||||
@@ -581,6 +583,65 @@ func TestScanContentHashedStalePartners(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentSameSecondRewrite is TestScanContentStalePartners for
|
||||
// a stored file rewritten in place at the same size with an mtime later
|
||||
// in the same second than recorded: the file counts as changed, so
|
||||
// neither it nor its match inside the operand is read.
|
||||
func TestScanContentSameSecondRewrite(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
db := openTestDB(t)
|
||||
dirA := t.TempDir()
|
||||
changed := sparseFileWithoutMatch(t, dirA, "changed", headTailMin)
|
||||
|
||||
first := time.Date(2026, 1, 2, 3, 4, 5, 100_000_000, time.UTC)
|
||||
|
||||
err := os.Chtimes(changed, first, first)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
syncTree(t, db, dirA)
|
||||
|
||||
before := dbRecords(t, db)
|
||||
|
||||
// Rewrite one byte in place, keeping the size.
|
||||
pokeAt(t, changed, headTailMin/2, []byte{1})
|
||||
|
||||
later := first.Add(500 * time.Millisecond)
|
||||
|
||||
err = os.Chtimes(changed, later, later)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
dirB := t.TempDir()
|
||||
sparseFile(t, dirB, "changed-copy", headTailMin)
|
||||
|
||||
st := syncTree(t, db, dirB)
|
||||
if st != (scanStats{walked: 1, added: 1}) {
|
||||
t.Errorf("stats = %+v, want 1 added and nothing skipped", st)
|
||||
}
|
||||
|
||||
recs := dbRecords(t, db)
|
||||
for _, r := range recs {
|
||||
if r.content != "" {
|
||||
t.Errorf("%s: content = %q, want none: its only match is stale",
|
||||
r.path, r.content)
|
||||
}
|
||||
}
|
||||
|
||||
for _, old := range before {
|
||||
if r := recordByPath(t, recs, old.path); r != old {
|
||||
t.Errorf("record = %+v, want it left as %+v", r, old)
|
||||
}
|
||||
}
|
||||
|
||||
if groups := dupeGroups(t, db); len(groups) != 0 {
|
||||
t.Errorf("groups = %+v, want none", groups)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScanContentReadFailure checks that a failed content read is
|
||||
// counted as skipped and leaves the record without a content hash, and
|
||||
// that a later scan tries the read again.
|
||||
@@ -783,8 +844,8 @@ func TestWalk(t *testing.T) {
|
||||
t.Errorf("%s: size = %d, want 1..3", r.path, r.size)
|
||||
}
|
||||
|
||||
if r.mtime <= 0 {
|
||||
t.Errorf("%s: mtime = %d, want positive", r.path, r.mtime)
|
||||
if r.mtime.Unix() <= 0 {
|
||||
t.Errorf("%s: mtime = %v, want after 1970", r.path, r.mtime)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1285,11 +1346,172 @@ func TestSyncScanMtimeBump(t *testing.T) {
|
||||
t.Fatalf("mtime-bump stats = %+v, want 1 updated", st)
|
||||
}
|
||||
|
||||
if r := recordByPath(t, dbRecords(t, db), a); r.mtime != future.Unix() {
|
||||
t.Fatalf("mtime = %d, want %d", r.mtime, future.Unix())
|
||||
if r := recordByPath(t, dbRecords(t, db), a); !r.mtime.Equal(future) {
|
||||
t.Fatalf("mtime = %v, want %v", r.mtime, future)
|
||||
}
|
||||
}
|
||||
|
||||
// assertWholeFileHashed fails unless the record for path holds the
|
||||
// whole-file hash of data as its head, tail, and content.
|
||||
func assertWholeFileHashed(t *testing.T, db *sql.DB, path string,
|
||||
data []byte,
|
||||
) {
|
||||
t.Helper()
|
||||
|
||||
r := recordByPath(t, dbRecords(t, db), path)
|
||||
if want := hexSum(data); r.head != want || r.tail != want ||
|
||||
r.content != want {
|
||||
t.Fatalf("head, tail, content = %q, %q, %q, want %q for each",
|
||||
r.head, r.tail, r.content, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSyncScanSameSecondRewrite rewrites a file in place at the same
|
||||
// size with an mtime later in the same second as the recorded one: the
|
||||
// next scan must notice the change and re-hash the file.
|
||||
func TestSyncScanSameSecondRewrite(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
db := openTestDB(t)
|
||||
a := writeFile(t, dir, "a.bin", pattern(1, 500))
|
||||
|
||||
// b.bin shares the size of a.bin, so a.bin is hashed.
|
||||
writeFile(t, dir, "b.bin", pattern(2, 500))
|
||||
|
||||
first := time.Date(2026, 1, 2, 3, 4, 5, 100_000_000, time.UTC)
|
||||
|
||||
err := os.Chtimes(a, first, first)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
syncTree(t, db, dir)
|
||||
|
||||
rewritten := pattern(3, 500)
|
||||
writeFile(t, dir, "a.bin", rewritten)
|
||||
|
||||
later := first.Add(500 * time.Millisecond)
|
||||
|
||||
err = os.Chtimes(a, later, later)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
st := syncTree(t, db, dir)
|
||||
if st != (scanStats{walked: 2, updated: 1, unchanged: 1}) {
|
||||
t.Fatalf("rescan stats = %+v, want 1 updated 1 unchanged", st)
|
||||
}
|
||||
|
||||
assertWholeFileHashed(t, db, a, rewritten)
|
||||
}
|
||||
|
||||
// TestSyncScanOperandSameSecondRewrite is TestSyncScanSameSecondRewrite
|
||||
// for files given to scan as operands, which scan stats without reading
|
||||
// their directory.
|
||||
func TestSyncScanOperandSameSecondRewrite(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
dir := t.TempDir()
|
||||
db := openTestDB(t)
|
||||
a := writeFile(t, dir, "a.bin", pattern(1, 500))
|
||||
|
||||
// b.bin shares the size of a.bin, so a.bin is hashed.
|
||||
b := writeFile(t, dir, "b.bin", pattern(2, 500))
|
||||
|
||||
first := time.Date(2026, 1, 2, 3, 4, 5, 100_000_000, time.UTC)
|
||||
|
||||
err := os.Chtimes(a, first, first)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
syncTree(t, db, a, b)
|
||||
|
||||
rewritten := pattern(3, 500)
|
||||
writeFile(t, dir, "a.bin", rewritten)
|
||||
|
||||
later := first.Add(500 * time.Millisecond)
|
||||
|
||||
err = os.Chtimes(a, later, later)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
st := syncTree(t, db, a, b)
|
||||
if st != (scanStats{walked: 2, updated: 1, unchanged: 1}) {
|
||||
t.Fatalf("rescan stats = %+v, want 1 updated 1 unchanged", st)
|
||||
}
|
||||
|
||||
assertWholeFileHashed(t, db, a, rewritten)
|
||||
}
|
||||
|
||||
// TestSyncScanRewriteAfter2262 runs assertLateRewriteRehashed with an
|
||||
// mtime after 2262, a time too late to count in nanoseconds in an int64.
|
||||
func TestSyncScanRewriteAfter2262(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
assertLateRewriteRehashed(t, time.Date(2300, 1, 2, 3, 4, 5, 0, time.UTC))
|
||||
}
|
||||
|
||||
// TestSyncScanRewritePastTimeLimit runs assertLateRewriteRehashed with an
|
||||
// mtime one second past the latest a time.Time holds without wrapping it
|
||||
// to a time far in the past.
|
||||
func TestSyncScanRewritePastTimeLimit(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
assertLateRewriteRehashed(t, time.Unix(9223371974719179008, 0))
|
||||
}
|
||||
|
||||
// assertLateRewriteRehashed scans a directory, rewrites a file in it in
|
||||
// place at the same size, sets its mtime to late, and fails unless the
|
||||
// next scan re-hashes the file. It skips where late does not fit the
|
||||
// platform's timespec or the filesystem does not store it.
|
||||
func assertLateRewriteRehashed(t *testing.T, late time.Time) {
|
||||
t.Helper()
|
||||
|
||||
dir := t.TempDir()
|
||||
db := openTestDB(t)
|
||||
a := writeFile(t, dir, "a.bin", pattern(1, 500))
|
||||
|
||||
// b.bin shares the size of a.bin, so a.bin is hashed.
|
||||
writeFile(t, dir, "b.bin", pattern(2, 500))
|
||||
|
||||
syncTree(t, db, dir)
|
||||
|
||||
rewritten := pattern(3, 500)
|
||||
writeFile(t, dir, "a.bin", rewritten)
|
||||
|
||||
// os.Chtimes cannot set such a time: it converts through UnixNano.
|
||||
ts, err := unix.TimeToTimespec(late)
|
||||
if err != nil {
|
||||
t.Skipf("an mtime %d seconds after 1970 does not fit this platform's "+
|
||||
"timespec: %v", late.Unix(), err)
|
||||
}
|
||||
|
||||
err = unix.UtimesNano(a, []unix.Timespec{ts, ts})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
fi, err := os.Lstat(a)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if fi.ModTime().Unix() != late.Unix() {
|
||||
t.Skipf("the filesystem stored the mtime as %d seconds after 1970, "+
|
||||
"not %d", fi.ModTime().Unix(), late.Unix())
|
||||
}
|
||||
|
||||
st := syncTree(t, db, dir)
|
||||
if st != (scanStats{walked: 2, updated: 1, unchanged: 1}) {
|
||||
t.Fatalf("rescan stats = %+v, want 1 updated 1 unchanged", st)
|
||||
}
|
||||
|
||||
assertWholeFileHashed(t, db, a, rewritten)
|
||||
}
|
||||
|
||||
func TestSyncScanAddRemove(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
@@ -1335,9 +1557,7 @@ func TestSyncScanSizeChange(t *testing.T) {
|
||||
|
||||
writeFile(t, dir, "f", pattern(1, 200))
|
||||
|
||||
mt := time.Unix(old.mtime, 0)
|
||||
|
||||
err := os.Chtimes(p, mt, mt)
|
||||
err := os.Chtimes(p, old.mtime, old.mtime)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
+3
-1
@@ -1,6 +1,8 @@
|
||||
#!/bin/sh
|
||||
# script/check: run all checks (test, lint, fmt-check). Our own
|
||||
# extension to scripts-to-rule-them-all. Must not modify any files.
|
||||
# extension to scripts-to-rule-them-all. test and lint are Docker
|
||||
# phases; fmt-check is native, because a formatter writes the working
|
||||
# tree. Must not modify any files.
|
||||
set -eu
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||
|
||||
+7
-5
@@ -14,15 +14,17 @@ main() {
|
||||
cd "$ROOT"
|
||||
"$SCRIPT_DIR/bootstrap"
|
||||
"$SCRIPT_DIR/check"
|
||||
# Own line: a failing command substitution inside an argument does
|
||||
# not trip `set -e`, so the inline form degrades silently to an
|
||||
# empty constant. The VERSION build argument takes precedence over
|
||||
# the version a build stage derives from the .git in the context.
|
||||
# The version and the tag each get their own line: a failing
|
||||
# command substitution inside an argument does not trip `set -e`,
|
||||
# so the inline form degrades silently to an empty constant. The
|
||||
# VERSION build argument takes precedence over the version a build
|
||||
# stage derives from the .git in the context.
|
||||
version="$(git describe --tags --always --dirty 2>/dev/null || true)"
|
||||
[ -n "$version" ] || version="unknown"
|
||||
tag="$("$SCRIPT_DIR/projectname")"
|
||||
docker build --no-cache \
|
||||
--build-arg VERSION="$version" \
|
||||
-t "$("$SCRIPT_DIR/projectname")" .
|
||||
-t "$tag" .
|
||||
}
|
||||
|
||||
main "$@"
|
||||
|
||||
+7
-5
@@ -10,15 +10,17 @@ ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
# Own line: a failing command substitution inside an argument does
|
||||
# not trip `set -e`, so the inline form degrades silently to an
|
||||
# empty constant. The VERSION build argument takes precedence over
|
||||
# the version a build stage derives from the .git in the context.
|
||||
# The version and the tag each get their own line: a failing
|
||||
# command substitution inside an argument does not trip `set -e`,
|
||||
# so the inline form degrades silently to an empty constant. The
|
||||
# VERSION build argument takes precedence over the version a build
|
||||
# stage derives from the .git in the context.
|
||||
version="$(git describe --tags --always --dirty 2>/dev/null || true)"
|
||||
[ -n "$version" ] || version="unknown"
|
||||
tag="$("$SCRIPT_DIR/projectname")"
|
||||
docker build --no-cache \
|
||||
--build-arg VERSION="$version" \
|
||||
-t "$("$SCRIPT_DIR/projectname")" .
|
||||
-t "$tag" .
|
||||
}
|
||||
|
||||
main "$@"
|
||||
|
||||
@@ -1,19 +1,15 @@
|
||||
#!/bin/sh
|
||||
# script/install-precommit: install the git pre-commit hook that runs
|
||||
# script/precommit. Our own extension to scripts-to-rule-them-all.
|
||||
# Hooks are shared between the main checkout and all worktrees, so
|
||||
# resolve the common git dir instead of assuming .git is a directory.
|
||||
set -eu
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
|
||||
main() {
|
||||
cd "$ROOT"
|
||||
hooks_dir="$(git rev-parse --git-common-dir)/hooks"
|
||||
mkdir -p "$hooks_dir"
|
||||
hook="$hooks_dir/pre-commit"
|
||||
printf '#!/bin/sh\nset -e\nscript/precommit\n' > "$hook"
|
||||
chmod +x "$hook"
|
||||
hook=".git/hooks/pre-commit"
|
||||
printf '#!/bin/sh\nset -e\nscript/precommit\n' > .git/hooks/pre-commit
|
||||
chmod +x .git/hooks/pre-commit
|
||||
echo "pre-commit hook installed: runs script/precommit"
|
||||
}
|
||||
|
||||
|
||||
+2
-2
@@ -1,13 +1,13 @@
|
||||
#!/bin/sh
|
||||
# script/precommit: run by the git pre-commit hook; fails the commit if
|
||||
# checks fail. Our own extension to scripts-to-rule-them-all. Go extra:
|
||||
# go mod tidy must be a no-op before the checks run.
|
||||
# checks fail. Our own extension to scripts-to-rule-them-all.
|
||||
set -eu
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
|
||||
|
||||
main() {
|
||||
# Go extra: go mod tidy must be a no-op before the checks run.
|
||||
cd "$ROOT"
|
||||
go mod tidy
|
||||
if ! git diff --exit-code -- go.mod go.sum; then
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
#!/bin/sh
|
||||
# script/setup: set up the repo for development after a fresh clone:
|
||||
# installs dependencies (script/bootstrap) and the git pre-commit hook.
|
||||
# installs dependencies and the git pre-commit hook.
|
||||
set -eu
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
|
||||
|
||||
Reference in New Issue
Block a user