Store mtime to the nanosecond so a same-second rewrite is re-hashed (closes #12)
check / check (push) Waiting to run

scan recorded mtime in whole seconds, so a file rewritten in place at
the same size within the same second as its recorded mtime was classed
unchanged and kept its old hashes. The files table keeps mtime as whole
Unix seconds and gains mtime_nsec, the nanoseconds within that second.
scan holds the mtime as a time.Time and decides "newer" with After, so
any time a filesystem can record, one after 2262 included, compares in
the right order. The walk, a file given as an operand, and the content
phase's recheck all move over. PRAGMA user_version stays 1, per the
owner's ruling. README states what both columns hold.

Model: opus-5-5
This commit is contained in:
2026-10-07 22:02:22 +00:00
committed by sneak
parent 0064eba542
commit 696565cb12
8 changed files with 308 additions and 68 deletions
+38 -28
View File
@@ -11,6 +11,7 @@ import (
"path/filepath"
"slices"
"strconv"
"time"
"golang.org/x/sys/unix"
// The pure-Go SQLite driver, registered as "sqlite"; keeps cgo
@@ -40,15 +41,18 @@ const dbDirPerm = 0o755
const lockFilePerm = 0o600
// createTableSQL is the schema applied to a fresh database. Paths are
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8.
// BLOBs because Unix paths are raw bytes, not guaranteed UTF-8. mtime
// holds whole Unix seconds and mtime_nsec the nanoseconds within that
// second.
const createTableSQL = `
CREATE TABLE files (
path BLOB PRIMARY KEY,
size INTEGER NOT NULL,
mtime INTEGER NOT NULL,
head TEXT NOT NULL,
tail TEXT NOT NULL,
content TEXT NOT NULL
path BLOB PRIMARY KEY,
size INTEGER NOT NULL,
mtime INTEGER NOT NULL,
mtime_nsec INTEGER NOT NULL,
head TEXT NOT NULL,
tail TEXT NOT NULL,
content TEXT NOT NULL
) WITHOUT ROWID
`
@@ -61,10 +65,11 @@ CREATE INDEX files_signature ON files (size, head, tail, content)
// upsertSQL inserts one file record, replacing any existing record for
// the same path.
const upsertSQL = `
INSERT INTO files (path, size, mtime, head, tail, content)
VALUES (?, ?, ?, ?, ?, ?)
INSERT INTO files (path, size, mtime, mtime_nsec, head, tail, content)
VALUES (?, ?, ?, ?, ?, ?, ?)
ON CONFLICT (path) DO UPDATE SET
size = excluded.size, mtime = excluded.mtime,
mtime_nsec = excluded.mtime_nsec,
head = excluded.head, tail = excluded.tail,
content = excluded.content
`
@@ -356,8 +361,8 @@ func userVersion(ctx context.Context, db *sql.DB) (int, error) {
// which is the order of the primary key, so SQLite does not sort.
func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
rows, err := db.QueryContext(ctx,
"SELECT path, size, mtime, head, tail, content FROM files "+
"ORDER BY path")
"SELECT path, size, mtime, mtime_nsec, head, tail, content "+
"FROM files ORDER BY path")
if err != nil {
return fmt.Errorf("read records: %w", err)
}
@@ -366,17 +371,19 @@ func loadFileRows(ctx context.Context, db *sql.DB, fn func(r scanRec)) error {
for rows.Next() {
var (
path []byte
r scanRec
path []byte
sec, nsec int64
r scanRec
)
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail,
err = rows.Scan(&path, &r.size, &sec, &nsec, &r.head, &r.tail,
&r.content)
if err != nil {
return fmt.Errorf("read record: %w", err)
}
r.path = string(path)
r.mtime = time.Unix(sec, nsec)
fn(r)
}
@@ -466,10 +473,10 @@ func loadDupeRows(ctx context.Context, db *sql.DB,
// values, and skipping the hash columns keeps the scan's in-memory
// index small on multi-million-file databases.
func loadFileMeta(ctx context.Context, db *sql.DB,
fn func(path string, size, mtime int64, hashed bool),
fn func(path string, size int64, mtime time.Time, hashed bool),
) error {
rows, err := db.QueryContext(ctx,
"SELECT path, size, mtime, head <> '' FROM files")
"SELECT path, size, mtime, mtime_nsec, head <> '' FROM files")
if err != nil {
return fmt.Errorf("read records: %w", err)
}
@@ -478,17 +485,17 @@ func loadFileMeta(ctx context.Context, db *sql.DB,
for rows.Next() {
var (
path []byte
size, mtime int64
hashed int64
path []byte
size, sec, nsec int64
hashed int64
)
err = rows.Scan(&path, &size, &mtime, &hashed)
err = rows.Scan(&path, &size, &sec, &nsec, &hashed)
if err != nil {
return fmt.Errorf("read record: %w", err)
}
fn(string(path), size, mtime, hashed != 0)
fn(string(path), size, time.Unix(sec, nsec), hashed != 0)
}
err = rows.Err()
@@ -507,7 +514,7 @@ func loadFileMeta(ctx context.Context, db *sql.DB,
// memory; the rows come ordered by size, head, and tail, so each
// group's rows arrive together.
const contentCandidatesSQL = `
SELECT f.path, f.size, f.mtime, f.head, f.tail, f.content <> ''
SELECT f.path, f.size, f.mtime, f.mtime_nsec, f.head, f.tail, f.content <> ''
FROM files AS f
JOIN (
SELECT size, head, tail
@@ -533,17 +540,20 @@ func loadContentCandidates(ctx context.Context, db *sql.DB,
for rows.Next() {
var (
path []byte
r scanRec
hashed int64
path []byte
sec, nsec int64
r scanRec
hashed int64
)
err = rows.Scan(&path, &r.size, &r.mtime, &r.head, &r.tail, &hashed)
err = rows.Scan(&path, &r.size, &sec, &nsec, &r.head, &r.tail,
&hashed)
if err != nil {
return fmt.Errorf("read record: %w", err)
}
r.path = string(path)
r.mtime = time.Unix(sec, nsec)
fn(r, hashed != 0)
}
@@ -627,8 +637,8 @@ func execUpserts(ctx context.Context, tx *sql.Tx, upserts []scanRec,
defer func() { _ = st.Close() }()
for _, r := range upserts {
_, err = st.ExecContext(ctx,
[]byte(r.path), r.size, r.mtime, r.head, r.tail, r.content)
_, err = st.ExecContext(ctx, []byte(r.path), r.size,
r.mtime.Unix(), r.mtime.Nanosecond(), r.head, r.tail, r.content)
if err != nil {
return fmt.Errorf("upsert %s: %w", r.path, err)
}