Store mtime to the nanosecond so a same-second rewrite is re-hashed (closes #12) #96
@@ -312,29 +312,34 @@ All three subcommands operate on a single SQLite database file:
|
|||||||
|
|
||||||
```sql
|
```sql
|
||||||
CREATE TABLE files (
|
CREATE TABLE files (
|
||||||
path BLOB PRIMARY KEY, -- absolute path, raw bytes
|
path BLOB PRIMARY KEY, -- absolute path, raw bytes
|
||||||
size INTEGER NOT NULL, -- bytes, from lstat
|
size INTEGER NOT NULL, -- bytes, from lstat
|
||||||
mtime INTEGER NOT NULL, -- Unix seconds, from lstat
|
mtime INTEGER NOT NULL, -- whole Unix seconds of the mtime, from lstat
|
||||||
head TEXT NOT NULL, -- lowercase-hex SHA-256; first 64 KiB, or whole file under 10 MiB
|
mtime_nsec INTEGER NOT NULL, -- nanoseconds within that second, 0 to 999999999
|
||||||
tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB
|
head TEXT NOT NULL, -- lowercase-hex SHA-256; first 64 KiB, or whole file under 10 MiB
|
||||||
content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples
|
tail TEXT NOT NULL, -- lowercase-hex SHA-256; last 64 KiB, or whole file under 10 MiB
|
||||||
|
content TEXT NOT NULL -- lowercase-hex SHA-256, whole file or samples
|
||||||
) WITHOUT ROWID;
|
) WITHOUT ROWID;
|
||||||
CREATE INDEX files_signature ON files (size, head, tail, content);
|
CREATE INDEX files_signature ON files (size, head, tail, content);
|
||||||
```
|
```
|
||||||
|
|
||||||
Paths are stored as BLOBs because Unix paths are raw bytes, not guaranteed
|
Paths are stored as BLOBs because Unix paths are raw bytes, not guaranteed
|
||||||
UTF-8. `mtime` is used only for change detection; it is not part of the
|
UTF-8. `mtime` and `mtime_nsec` hold the file's mtime to the nanosecond:
|
||||||
duplicate key. For a file under 10 MiB `head`, `tail`, and `content` all
|
`mtime` the whole Unix seconds, rounded down, and `mtime_nsec` the
|
||||||
hold the whole-file hash (that range is hashed in full, with no end
|
nanoseconds past that second. Split this way they hold any mtime a
|
||||||
windows); for a larger file `head` and `tail` hold the first- and last-64
|
filesystem can record, one before 1678 or after 2262 included, which a
|
||||||
KiB hashes and `content` the whole-file or sampled hash. All three are empty
|
single 64-bit count of nanoseconds cannot. They are used only for change
|
||||||
strings when the file has never been hashed because its size was unique as
|
detection and are not part of the duplicate key. For a file under 10 MiB
|
||||||
of the last scan that covered it. For a file of 10 MiB or more, `content`
|
`head`, `tail`, and `content` all hold the whole-file hash (that range is
|
||||||
stays empty until the content phase of a scan (see "`scan` mode" below) has
|
hashed in full, with no end windows); for a larger file `head` and `tail`
|
||||||
read the file. A record with an empty `content` is never part of a duplicate
|
hold the first- and last-64 KiB hashes and `content` the whole-file or
|
||||||
group, though it still defines the file for tree reconstruction. The
|
sampled hash. All three are empty strings when the file has never been
|
||||||
`files_signature` index lets SQLite group the records by signature for
|
hashed because its size was unique as of the last scan that covered it. For
|
||||||
`report` without sorting the whole table.
|
a file of 10 MiB or more, `content` stays empty until the content phase of a
|
||||||
|
scan (see "`scan` mode" below) has read the file. A record with an empty
|
||||||
|
`content` is never part of a duplicate group, though it still defines the
|
||||||
|
file for tree reconstruction. The `files_signature` index lets SQLite group
|
||||||
|
the records by signature for `report` without sorting the whole table.
|
||||||
|
|
||||||
### Duplicate detection
|
### Duplicate detection
|
||||||
|
|
||||||
@@ -421,6 +426,9 @@ operands:
|
|||||||
once its size, `head`, and `tail` match another record's.
|
once its size, `head`, and `tail` match another record's.
|
||||||
- A file whose mtime is newer than recorded, or whose size differs, is processed
|
- A file whose mtime is newer than recorded, or whose size differs, is processed
|
||||||
as if new: re-hashed, or recorded without hashes, per the shared-size rule.
|
as if new: re-hashed, or recorded without hashes, per the shared-size rule.
|
||||||
|
Change detection compares the mtime to the nanosecond, as finely as the
|
||||||
|
filesystem records it, so a same-size rewrite counts as a change whenever the
|
||||||
|
filesystem gives it a later mtime than recorded, even within the same second.
|
||||||
- A database record whose path lies under one of the scanned operands but was
|
- A database record whose path lies under one of the scanned operands but was
|
||||||
not successfully processed this run is deleted. This removes records for
|
not successfully processed this run is deleted. This removes records for
|
||||||
deleted files. It also removes records for paths that failed to stat or hash
|
deleted files. It also removes records for paths that failed to stat or hash
|
||||||
|
|||||||
@@ -29,6 +29,11 @@
|
|||||||
|
|
||||||
# Completed Steps
|
# Completed Steps
|
||||||
|
|
||||||
|
- `scan` records mtime to the nanosecond, as whole seconds in `mtime` plus
|
||||||
|
`mtime_nsec`, and compares it at that resolution, so a same-size rewrite
|
||||||
|
within the same second is re-hashed (2026-10-07,
|
||||||
|
https://git.eeqj.de/sneak/sfdupes/issues/12)
|
||||||
|
|
||||||
- cut the narration from `TODO.md` Completed Steps and from the comments in
|
- cut the narration from `TODO.md` Completed Steps and from the comments in
|
||||||
`script/` and both Dockerfiles; §Workflow now branches from and merges to
|
`script/` and both Dockerfiles; §Workflow now branches from and merges to
|
||||||
`next` (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/49)
|
`next` (2026-10-04, https://git.eeqj.de/sneak/sfdupes/issues/49)
|
||||||
|
|||||||
@@ -11,6 +11,7 @@ import (
|
|||||||
"path/filepath"
|
"path/filepath"
|
||||||
"slices"
|
"slices"
|
||||||
"strconv"
|
"strconv"
|
||||||
|
"time"
|
||||||
|
|
||||||
"golang.org/x/sys/unix"
|
"golang.org/x/sys/unix"
|
||||||
// The pure-Go SQLite driver, registered as "sqlite"; keeps cgo
|
// The pure-Go SQLite driver, registered as "sqlite"; keeps cgo
|
||||||
@@ -40,15 +41,18 @@ const dbDirPerm = 0o755
|
|||||||
const lockFilePerm = 0o600
|
const lockFilePerm = 0o600
|
||||||
|
|
||||||
// createTableSQL is the schema applied to a fresh database. Paths are
|
// createTableSQL is the schema applied to a fresh database. Paths are
|
||||||
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8.
|
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8. mtime
|
||||||
|
// holds whole Unix seconds and mtime_nsec the nanoseconds within that
|
||||||
|
// second.
|
||||||
const createTableSQL = `
|
const createTableSQL = `
|
||||||
CREATE TABLE files (
|
CREATE TABLE files (
|
||||||
path BLOB PRIMARY KEY,
|
path BLOB PRIMARY KEY,
|
||||||
size INTEGER NOT NULL,
|
size INTEGER NOT NULL,
|
||||||
mtime INTEGER NOT NULL,
|
mtime INTEGER NOT NULL,
|
||||||
head TEXT NOT NULL,
|
mtime_nsec INTEGER NOT NULL,
|
||||||
tail TEXT NOT NULL,
|
head TEXT NOT NULL,
|
||||||
content TEXT NOT NULL
|
tail TEXT NOT NULL,
|
||||||
|
content TEXT NOT NULL
|
||||||
) WITHOUT ROWID
|
) WITHOUT ROWID
|
||||||
`
|
`
|
||||||
|
|
||||||
@@ -61,10 +65,11 @@ CREATE INDEX files_signature ON files (size, head, tail, content)
|
|||||||
// upsertSQL inserts one file record, replacing any existing record for
|
// upsertSQL inserts one file record, replacing any existing record for
|
||||||
// the same path.
|
// the same path.
|
||||||
const upsertSQL = `
|
const upsertSQL = `
|
||||||
INSERT INTO files (path, size, mtime, head, tail, content)
|
INSERT INTO files (path, size, mtime, mtime_nsec, head, tail, content)
|
||||||
VALUES (?, ?, ?, ?, ?, ?)
|
VALUES (?, ?, ?, ?, ?, ?, ?)
|
||||||
ON CONFLICT (path) DO UPDATE SET
|
ON CONFLICT (path) DO UPDATE SET
|
||||||
size = excluded.size, mtime = excluded.mtime,
|
size = excluded.size, mtime = excluded.mtime,
|
||||||
|
mtime_nsec = excluded.mtime_nsec,
|
||||||
head = excluded.head, tail = excluded.tail,
|
head = excluded.head, tail = excluded.tail,
|
||||||
content = excluded.content
|
content = excluded.content
|
||||||
`
|
`
|
||||||
@@ -356,8 +361,8 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) {
|
|||||||
// which is the order of the primary key, so SQLite does not sort.
|
// which is the order of the primary key, so SQLite does not sort.
|
||||||
func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
|
func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
|
||||||
rows, err := db.QueryContext(ctx,
|
rows, err := db.QueryContext(ctx,
|
||||||
"SELECT path, size, mtime, head, tail, content FROM files "+
|
"SELECT path, size, mtime, mtime_nsec, head, tail, content "+
|
||||||
"ORDER BY path")
|
"FROM files ORDER BY path")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read records: %w", err)
|
return fmt.Errorf("read records: %w", err)
|
||||||
}
|
}
|
||||||
@@ -366,17 +371,19 @@ func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
|
|||||||
|
|
||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
var (
|
var (
|
||||||
path []byte
|
path []byte
|
||||||
r scanRec
|
sec, nsec int64
|
||||||
|
r scanRec
|
||||||
)
|
)
|
||||||
|
|
||||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
|
err = rows.Scan(&path, &r.size, &sec, &nsec, &r.head, &r.tail,
|
||||||
&r.content)
|
&r.content)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read record: %w", err)
|
return fmt.Errorf("read record: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
r.path = string(path)
|
r.path = string(path)
|
||||||
|
r.mtime = time.Unix(sec, nsec)
|
||||||
fn(r)
|
fn(r)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -466,10 +473,10 @@ func loadDupeRows(ctx context.Context, db *sql.DB,
|
|||||||
// values, and skipping the hash columns keeps the scan's in-memory
|
// values, and skipping the hash columns keeps the scan's in-memory
|
||||||
// index small on multi-million-file databases.
|
// index small on multi-million-file databases.
|
||||||
func loadFileMeta(ctx context.Context, db *sql.DB,
|
func loadFileMeta(ctx context.Context, db *sql.DB,
|
||||||
fn func(path string, size, mtime int64, hashed bool),
|
fn func(path string, size int64, mtime time.Time, hashed bool),
|
||||||
) error {
|
) error {
|
||||||
rows, err := db.QueryContext(ctx,
|
rows, err := db.QueryContext(ctx,
|
||||||
"SELECT path, size, mtime, head <> '' FROM files")
|
"SELECT path, size, mtime, mtime_nsec, head <> '' FROM files")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read records: %w", err)
|
return fmt.Errorf("read records: %w", err)
|
||||||
}
|
}
|
||||||
@@ -478,17 +485,17 @@ func loadFileMeta(ctx context.Context, db *sql.DB,
|
|||||||
|
|
||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
var (
|
var (
|
||||||
path []byte
|
path []byte
|
||||||
size, mtime int64
|
size, sec, nsec int64
|
||||||
hashed int64
|
hashed int64
|
||||||
)
|
)
|
||||||
|
|
||||||
err = rows.Scan(&path, &size, &mtime, &hashed)
|
err = rows.Scan(&path, &size, &sec, &nsec, &hashed)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read record: %w", err)
|
return fmt.Errorf("read record: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn(string(path), size, mtime, hashed != 0)
|
fn(string(path), size, time.Unix(sec, nsec), hashed != 0)
|
||||||
}
|
}
|
||||||
|
|
||||||
err = rows.Err()
|
err = rows.Err()
|
||||||
@@ -507,7 +514,7 @@ func loadFileMeta(ctx context.Context, db *sql.DB,
|
|||||||
// memory; the rows come ordered by size, head, and tail, so each
|
// memory; the rows come ordered by size, head, and tail, so each
|
||||||
// group's rows arrive together.
|
// group's rows arrive together.
|
||||||
const contentCandidatesSQL = `
|
const contentCandidatesSQL = `
|
||||||
SELECT f.path, f.size, f.mtime, f.head, f.tail, f.content <> ''
|
SELECT f.path, f.size, f.mtime, f.mtime_nsec, f.head, f.tail, f.content <> ''
|
||||||
FROM files AS f
|
FROM files AS f
|
||||||
JOIN (
|
JOIN (
|
||||||
SELECT size, head, tail
|
SELECT size, head, tail
|
||||||
@@ -533,17 +540,20 @@ func loadContentCandidates(ctx context.Context, db *sql.DB,
|
|||||||
|
|
||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
var (
|
var (
|
||||||
path []byte
|
path []byte
|
||||||
r scanRec
|
sec, nsec int64
|
||||||
hashed int64
|
r scanRec
|
||||||
|
hashed int64
|
||||||
)
|
)
|
||||||
|
|
||||||
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, &hashed)
|
err = rows.Scan(&path, &r.size, &sec, &nsec, &r.head, &r.tail,
|
||||||
|
&hashed)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("read record: %w", err)
|
return fmt.Errorf("read record: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
r.path = string(path)
|
r.path = string(path)
|
||||||
|
r.mtime = time.Unix(sec, nsec)
|
||||||
fn(r, hashed != 0)
|
fn(r, hashed != 0)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -627,8 +637,8 @@ func execUpserts(ctx context.Context, tx *sql.Tx, upserts []scanRec,
|
|||||||
defer func() { _ = st.Close() }()
|
defer func() { _ = st.Close() }()
|
||||||
|
|
||||||
for _, r := range upserts {
|
for _, r := range upserts {
|
||||||
_, err = st.ExecContext(ctx,
|
_, err = st.ExecContext(ctx, []byte(r.path), r.size,
|
||||||
[]byte(r.path), r.size, r.mtime, r.head, r.tail, r.content)
|
r.mtime.Unix(), r.mtime.Nanosecond(), r.head, r.tail, r.content)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("upsert %s: %w", r.path, err)
|
return fmt.Errorf("upsert %s: %w", r.path, err)
|
||||||
}
|
}
|
||||||
|
|||||||
+10
-5
@@ -10,6 +10,7 @@ import (
|
|||||||
"slices"
|
"slices"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
"time"
|
||||||
)
|
)
|
||||||
|
|
||||||
// testDBPath returns a database path inside a fresh temp dir.
|
// testDBPath returns a database path inside a fresh temp dir.
|
||||||
@@ -260,10 +261,13 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
|||||||
// written.
|
// written.
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{
|
{
|
||||||
size: 2, mtime: 20, head: "h2", tail: "t2", content: "c2",
|
size: 2, mtime: time.Unix(20, 999_999_999), head: "h2", tail: "t2",
|
||||||
path: "/a/tab\tnew\nline",
|
content: "c2", path: "/a/tab\tnew\nline",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
size: 1, mtime: time.Unix(10, 0), head: "h1", tail: "t1",
|
||||||
|
content: "c1", path: "/a/x",
|
||||||
},
|
},
|
||||||
{size: 1, mtime: 10, head: "h1", tail: "t1", content: "c1", path: "/a/x"},
|
|
||||||
}
|
}
|
||||||
|
|
||||||
err := applyChanges(t.Context(), db, recs, nil,
|
err := applyChanges(t.Context(), db, recs, nil,
|
||||||
@@ -281,7 +285,8 @@ func TestApplyChangesRoundTrip(t *testing.T) {
|
|||||||
// An upsert for an existing path updates in place; a delete
|
// An upsert for an existing path updates in place; a delete
|
||||||
// removes exactly its path.
|
// removes exactly its path.
|
||||||
upd := scanRec{
|
upd := scanRec{
|
||||||
size: 3, mtime: 30, head: "h3", tail: "t3", content: "c3", path: "/a/x",
|
size: 3, mtime: time.Unix(30, 0), head: "h3", tail: "t3", content: "c3",
|
||||||
|
path: "/a/x",
|
||||||
}
|
}
|
||||||
|
|
||||||
err = applyChanges(t.Context(), db, []scanRec{upd},
|
err = applyChanges(t.Context(), db, []scanRec{upd},
|
||||||
@@ -308,7 +313,7 @@ func TestApplyChangesBatching(t *testing.T) {
|
|||||||
recs := make([]scanRec, 0, n)
|
recs := make([]scanRec, 0, n)
|
||||||
for i := range n {
|
for i := range n {
|
||||||
recs = append(recs, scanRec{
|
recs = append(recs, scanRec{
|
||||||
size: int64(i), mtime: 1, head: "h", tail: "t",
|
size: int64(i), mtime: time.Unix(1, 0), head: "h", tail: "t",
|
||||||
path: fmt.Sprintf("/batch/%07d", i),
|
path: fmt.Sprintf("/batch/%07d", i),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ import (
|
|||||||
"io"
|
"io"
|
||||||
"os"
|
"os"
|
||||||
"strings"
|
"strings"
|
||||||
|
"time"
|
||||||
)
|
)
|
||||||
|
|
||||||
// ioBufSize is the buffer size for the buffered stdout writers.
|
// ioBufSize is the buffer size for the buffered stdout writers.
|
||||||
@@ -21,7 +22,7 @@ const minGroupSize = 2
|
|||||||
// only and used by scan for change detection.
|
// only and used by scan for change detection.
|
||||||
type scanRec struct {
|
type scanRec struct {
|
||||||
size int64
|
size int64
|
||||||
mtime int64
|
mtime time.Time
|
||||||
head string
|
head string
|
||||||
tail string
|
tail string
|
||||||
content string
|
content string
|
||||||
|
|||||||
+9
-2
@@ -11,6 +11,7 @@ import (
|
|||||||
"slices"
|
"slices"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
"time"
|
||||||
)
|
)
|
||||||
|
|
||||||
// awkwardDir is a directory name holding every byte the reports escape.
|
// awkwardDir is a directory name holding every byte the reports escape.
|
||||||
@@ -313,8 +314,14 @@ func TestDupeGroupsMtimeExcluded(t *testing.T) {
|
|||||||
// mtime is informational only; records differing only in mtime
|
// mtime is informational only; records differing only in mtime
|
||||||
// still group together.
|
// still group together.
|
||||||
recs := []scanRec{
|
recs := []scanRec{
|
||||||
{size: 9, mtime: 100, head: "h", tail: "t", content: "c", path: "/m/1"},
|
{
|
||||||
{size: 9, mtime: 200, head: "h", tail: "t", content: "c", path: "/m/2"},
|
size: 9, mtime: time.Unix(100, 0), head: "h", tail: "t",
|
||||||
|
content: "c", path: "/m/1",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
size: 9, mtime: time.Unix(200, 0), head: "h", tail: "t",
|
||||||
|
content: "c", path: "/m/2",
|
||||||
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
groups := dupeGroupsOf(t, recs)
|
groups := dupeGroupsOf(t, recs)
|
||||||
|
|||||||
@@ -17,6 +17,7 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"sync"
|
"sync"
|
||||||
"syscall"
|
"syscall"
|
||||||
|
"time"
|
||||||
)
|
)
|
||||||
|
|
||||||
// The duplicate ladder (see hashSignature and README "Duplicate
|
// The duplicate ladder (see hashSignature and README "Duplicate
|
||||||
@@ -68,7 +69,7 @@ var errInterrupted = errors.New("scan interrupted")
|
|||||||
type fileRec struct {
|
type fileRec struct {
|
||||||
path string
|
path string
|
||||||
size int64
|
size int64
|
||||||
mtime int64
|
mtime time.Time
|
||||||
dev uint64
|
dev uint64
|
||||||
ino uint64
|
ino uint64
|
||||||
}
|
}
|
||||||
@@ -79,7 +80,7 @@ type fileRec struct {
|
|||||||
// they would dominate the scan's memory.
|
// they would dominate the scan's memory.
|
||||||
type fileMeta struct {
|
type fileMeta struct {
|
||||||
size int64
|
size int64
|
||||||
mtime int64
|
mtime time.Time
|
||||||
hashed bool
|
hashed bool
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -369,7 +370,7 @@ func (s *scanState) loadIndex(ctx context.Context, roots []string) error {
|
|||||||
s.existing = make(map[string]fileMeta)
|
s.existing = make(map[string]fileMeta)
|
||||||
|
|
||||||
return loadFileMeta(ctx, s.db,
|
return loadFileMeta(ctx, s.db,
|
||||||
func(path string, size, mtime int64, hashed bool) {
|
func(path string, size int64, mtime time.Time, hashed bool) {
|
||||||
prog.increment()
|
prog.increment()
|
||||||
|
|
||||||
if underAnyRoot(path, roots) {
|
if underAnyRoot(path, roots) {
|
||||||
@@ -415,7 +416,7 @@ func (s *scanState) walkPhase(
|
|||||||
old, ok := s.existing[ev.rec.path]
|
old, ok := s.existing[ev.rec.path]
|
||||||
|
|
||||||
switch {
|
switch {
|
||||||
case !ok || old.size != ev.rec.size || old.mtime < ev.rec.mtime:
|
case !ok || old.size != ev.rec.size || mtimeAfter(ev.rec.mtime, old.mtime):
|
||||||
changed = append(changed, ev.rec)
|
changed = append(changed, ev.rec)
|
||||||
case old.hashed:
|
case old.hashed:
|
||||||
delete(s.existing, ev.rec.path)
|
delete(s.existing, ev.rec.path)
|
||||||
@@ -807,7 +808,7 @@ func unchangedFile(r scanRec) (fileRec, bool, error) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if !fi.Mode().IsRegular() || fi.Size() != r.size ||
|
if !fi.Mode().IsRegular() || fi.Size() != r.size ||
|
||||||
fi.ModTime().Unix() > r.mtime {
|
mtimeAfter(fi.ModTime(), r.mtime) {
|
||||||
return fileRec{}, false, nil
|
return fileRec{}, false, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -818,6 +819,16 @@ func unchangedFile(r scanRec) (fileRec, bool, error) {
|
|||||||
}, true, nil
|
}, true, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// mtimeAfter reports whether mtime a is later than mtime b.
|
||||||
|
// Not a.After(b): time.Time wraps an mtime past year 292 billion; Unix() undoes it.
|
||||||
|
func mtimeAfter(a, b time.Time) bool {
|
||||||
|
if a.Unix() != b.Unix() {
|
||||||
|
return a.Unix() > b.Unix()
|
||||||
|
}
|
||||||
|
|
||||||
|
return a.Nanosecond() > b.Nanosecond()
|
||||||
|
}
|
||||||
|
|
||||||
// underAnyRoot reports whether path is any of the roots or lies under
|
// underAnyRoot reports whether path is any of the roots or lies under
|
||||||
// one of them.
|
// one of them.
|
||||||
func underAnyRoot(path string, roots []string) bool {
|
func underAnyRoot(path string, roots []string) bool {
|
||||||
@@ -966,7 +977,7 @@ func seedRoot(ctx context.Context, root string,
|
|||||||
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
||||||
path: root,
|
path: root,
|
||||||
size: fi.Size(),
|
size: fi.Size(),
|
||||||
mtime: fi.ModTime().Unix(),
|
mtime: fi.ModTime(),
|
||||||
dev: dev,
|
dev: dev,
|
||||||
ino: ino,
|
ino: ino,
|
||||||
}})
|
}})
|
||||||
@@ -1124,7 +1135,7 @@ func emitFile(ctx context.Context, p string, e fs.DirEntry,
|
|||||||
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
sendEvent(ctx, events, walkEvent{rec: fileRec{
|
||||||
path: p,
|
path: p,
|
||||||
size: info.Size(),
|
size: info.Size(),
|
||||||
mtime: info.ModTime().Unix(),
|
mtime: info.ModTime(),
|
||||||
dev: dev,
|
dev: dev,
|
||||||
ino: ino,
|
ino: ino,
|
||||||
}})
|
}})
|
||||||
|
|||||||
+227
-7
@@ -20,6 +20,8 @@ import (
|
|||||||
"syscall"
|
"syscall"
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
|
"golang.org/x/sys/unix"
|
||||||
)
|
)
|
||||||
|
|
||||||
// writeFile creates a file with the given content and returns its path.
|
// writeFile creates a file with the given content and returns its path.
|
||||||
@@ -581,6 +583,65 @@ func TestScanContentHashedStalePartners(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestScanContentSameSecondRewrite is TestScanContentStalePartners for
|
||||||
|
// a stored file rewritten in place at the same size with an mtime later
|
||||||
|
// in the same second than recorded: the file counts as changed, so
|
||||||
|
// neither it nor its match inside the operand is read.
|
||||||
|
func TestScanContentSameSecondRewrite(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
db := openTestDB(t)
|
||||||
|
dirA := t.TempDir()
|
||||||
|
changed := sparseFileWithoutMatch(t, dirA, "changed", headTailMin)
|
||||||
|
|
||||||
|
first := time.Date(2026, 1, 2, 3, 4, 5, 100_000_000, time.UTC)
|
||||||
|
|
||||||
|
err := os.Chtimes(changed, first, first)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
syncTree(t, db, dirA)
|
||||||
|
|
||||||
|
before := dbRecords(t, db)
|
||||||
|
|
||||||
|
// Rewrite one byte in place, keeping the size.
|
||||||
|
pokeAt(t, changed, headTailMin/2, []byte{1})
|
||||||
|
|
||||||
|
later := first.Add(500 * time.Millisecond)
|
||||||
|
|
||||||
|
err = os.Chtimes(changed, later, later)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
dirB := t.TempDir()
|
||||||
|
sparseFile(t, dirB, "changed-copy", headTailMin)
|
||||||
|
|
||||||
|
st := syncTree(t, db, dirB)
|
||||||
|
if st != (scanStats{walked: 1, added: 1}) {
|
||||||
|
t.Errorf("stats = %+v, want 1 added and nothing skipped", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
recs := dbRecords(t, db)
|
||||||
|
for _, r := range recs {
|
||||||
|
if r.content != "" {
|
||||||
|
t.Errorf("%s: content = %q, want none: its only match is stale",
|
||||||
|
r.path, r.content)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, old := range before {
|
||||||
|
if r := recordByPath(t, recs, old.path); r != old {
|
||||||
|
t.Errorf("record = %+v, want it left as %+v", r, old)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if groups := dupeGroups(t, db); len(groups) != 0 {
|
||||||
|
t.Errorf("groups = %+v, want none", groups)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestScanContentReadFailure checks that a failed content read is
|
// TestScanContentReadFailure checks that a failed content read is
|
||||||
// counted as skipped and leaves the record without a content hash, and
|
// counted as skipped and leaves the record without a content hash, and
|
||||||
// that a later scan tries the read again.
|
// that a later scan tries the read again.
|
||||||
@@ -783,8 +844,8 @@ func TestWalk(t *testing.T) {
|
|||||||
t.Errorf("%s: size = %d, want 1..3", r.path, r.size)
|
t.Errorf("%s: size = %d, want 1..3", r.path, r.size)
|
||||||
}
|
}
|
||||||
|
|
||||||
if r.mtime <= 0 {
|
if r.mtime.Unix() <= 0 {
|
||||||
t.Errorf("%s: mtime = %d, want positive", r.path, r.mtime)
|
t.Errorf("%s: mtime = %v, want after 1970", r.path, r.mtime)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1285,11 +1346,172 @@ func TestSyncScanMtimeBump(t *testing.T) {
|
|||||||
t.Fatalf("mtime-bump stats = %+v, want 1 updated", st)
|
t.Fatalf("mtime-bump stats = %+v, want 1 updated", st)
|
||||||
}
|
}
|
||||||
|
|
||||||
if r := recordByPath(t, dbRecords(t, db), a); r.mtime != future.Unix() {
|
if r := recordByPath(t, dbRecords(t, db), a); !r.mtime.Equal(future) {
|
||||||
t.Fatalf("mtime = %d, want %d", r.mtime, future.Unix())
|
t.Fatalf("mtime = %v, want %v", r.mtime, future)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// assertWholeFileHashed fails unless the record for path holds the
|
||||||
|
// whole-file hash of data as its head, tail, and content.
|
||||||
|
func assertWholeFileHashed(t *testing.T, db *sql.DB, path string,
|
||||||
|
data []byte,
|
||||||
|
) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
r := recordByPath(t, dbRecords(t, db), path)
|
||||||
|
if want := hexSum(data); r.head != want || r.tail != want ||
|
||||||
|
r.content != want {
|
||||||
|
t.Fatalf("head, tail, content = %q, %q, %q, want %q for each",
|
||||||
|
r.head, r.tail, r.content, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSyncScanSameSecondRewrite rewrites a file in place at the same
|
||||||
|
// size with an mtime later in the same second as the recorded one: the
|
||||||
|
// next scan must notice the change and re-hash the file.
|
||||||
|
func TestSyncScanSameSecondRewrite(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
db := openTestDB(t)
|
||||||
|
a := writeFile(t, dir, "a.bin", pattern(1, 500))
|
||||||
|
|
||||||
|
// b.bin shares the size of a.bin, so a.bin is hashed.
|
||||||
|
writeFile(t, dir, "b.bin", pattern(2, 500))
|
||||||
|
|
||||||
|
first := time.Date(2026, 1, 2, 3, 4, 5, 100_000_000, time.UTC)
|
||||||
|
|
||||||
|
err := os.Chtimes(a, first, first)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
syncTree(t, db, dir)
|
||||||
|
|
||||||
|
rewritten := pattern(3, 500)
|
||||||
|
writeFile(t, dir, "a.bin", rewritten)
|
||||||
|
|
||||||
|
later := first.Add(500 * time.Millisecond)
|
||||||
|
|
||||||
|
err = os.Chtimes(a, later, later)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
st := syncTree(t, db, dir)
|
||||||
|
if st != (scanStats{walked: 2, updated: 1, unchanged: 1}) {
|
||||||
|
t.Fatalf("rescan stats = %+v, want 1 updated 1 unchanged", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
assertWholeFileHashed(t, db, a, rewritten)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSyncScanOperandSameSecondRewrite is TestSyncScanSameSecondRewrite
|
||||||
|
// for files given to scan as operands, which scan stats without reading
|
||||||
|
// their directory.
|
||||||
|
func TestSyncScanOperandSameSecondRewrite(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
db := openTestDB(t)
|
||||||
|
a := writeFile(t, dir, "a.bin", pattern(1, 500))
|
||||||
|
|
||||||
|
// b.bin shares the size of a.bin, so a.bin is hashed.
|
||||||
|
b := writeFile(t, dir, "b.bin", pattern(2, 500))
|
||||||
|
|
||||||
|
first := time.Date(2026, 1, 2, 3, 4, 5, 100_000_000, time.UTC)
|
||||||
|
|
||||||
|
err := os.Chtimes(a, first, first)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
syncTree(t, db, a, b)
|
||||||
|
|
||||||
|
rewritten := pattern(3, 500)
|
||||||
|
writeFile(t, dir, "a.bin", rewritten)
|
||||||
|
|
||||||
|
later := first.Add(500 * time.Millisecond)
|
||||||
|
|
||||||
|
err = os.Chtimes(a, later, later)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
st := syncTree(t, db, a, b)
|
||||||
|
if st != (scanStats{walked: 2, updated: 1, unchanged: 1}) {
|
||||||
|
t.Fatalf("rescan stats = %+v, want 1 updated 1 unchanged", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
assertWholeFileHashed(t, db, a, rewritten)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSyncScanRewriteAfter2262 runs assertLateRewriteRehashed with an
|
||||||
|
// mtime after 2262, a time too late to count in nanoseconds in an int64.
|
||||||
|
func TestSyncScanRewriteAfter2262(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
assertLateRewriteRehashed(t, time.Date(2300, 1, 2, 3, 4, 5, 0, time.UTC))
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSyncScanRewritePastTimeLimit runs assertLateRewriteRehashed with an
|
||||||
|
// mtime one second past the latest a time.Time holds without wrapping it
|
||||||
|
// to a time far in the past.
|
||||||
|
func TestSyncScanRewritePastTimeLimit(t *testing.T) {
|
||||||
|
t.Parallel()
|
||||||
|
|
||||||
|
assertLateRewriteRehashed(t, time.Unix(9223371974719179008, 0))
|
||||||
|
}
|
||||||
|
|
||||||
|
// assertLateRewriteRehashed scans a directory, rewrites a file in it in
|
||||||
|
// place at the same size, sets its mtime to late, and fails unless the
|
||||||
|
// next scan re-hashes the file. It skips where late does not fit the
|
||||||
|
// platform's timespec or the filesystem does not store it.
|
||||||
|
func assertLateRewriteRehashed(t *testing.T, late time.Time) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
dir := t.TempDir()
|
||||||
|
db := openTestDB(t)
|
||||||
|
a := writeFile(t, dir, "a.bin", pattern(1, 500))
|
||||||
|
|
||||||
|
// b.bin shares the size of a.bin, so a.bin is hashed.
|
||||||
|
writeFile(t, dir, "b.bin", pattern(2, 500))
|
||||||
|
|
||||||
|
syncTree(t, db, dir)
|
||||||
|
|
||||||
|
rewritten := pattern(3, 500)
|
||||||
|
writeFile(t, dir, "a.bin", rewritten)
|
||||||
|
|
||||||
|
// os.Chtimes cannot set such a time: it converts through UnixNano.
|
||||||
|
ts, err := unix.TimeToTimespec(late)
|
||||||
|
if err != nil {
|
||||||
|
t.Skipf("an mtime %d seconds after 1970 does not fit this platform's "+
|
||||||
|
"timespec: %v", late.Unix(), err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = unix.UtimesNano(a, []unix.Timespec{ts, ts})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
fi, err := os.Lstat(a)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if fi.ModTime().Unix() != late.Unix() {
|
||||||
|
t.Skipf("the filesystem stored the mtime as %d seconds after 1970, "+
|
||||||
|
"not %d", fi.ModTime().Unix(), late.Unix())
|
||||||
|
}
|
||||||
|
|
||||||
|
st := syncTree(t, db, dir)
|
||||||
|
if st != (scanStats{walked: 2, updated: 1, unchanged: 1}) {
|
||||||
|
t.Fatalf("rescan stats = %+v, want 1 updated 1 unchanged", st)
|
||||||
|
}
|
||||||
|
|
||||||
|
assertWholeFileHashed(t, db, a, rewritten)
|
||||||
|
}
|
||||||
|
|
||||||
func TestSyncScanAddRemove(t *testing.T) {
|
func TestSyncScanAddRemove(t *testing.T) {
|
||||||
t.Parallel()
|
t.Parallel()
|
||||||
|
|
||||||
@@ -1335,9 +1557,7 @@ func TestSyncScanSizeChange(t *testing.T) {
|
|||||||
|
|
||||||
writeFile(t, dir, "f", pattern(1, 200))
|
writeFile(t, dir, "f", pattern(1, 200))
|
||||||
|
|
||||||
mt := time.Unix(old.mtime, 0)
|
err := os.Chtimes(p, old.mtime, old.mtime)
|
||||||
|
|
||||||
err := os.Chtimes(p, mt, mt)
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatal(err)
|
t.Fatal(err)
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user