Store and compare file mtimes to the nanosecond #257

Merged
clawbot merged 1 commits from issue-226-nanosecond-mtime into next 2026-10-07 07:12:13 +02:00
8 changed files with 226 additions and 24 deletions
+10
View File
@@ -22,6 +22,16 @@ the tag exists and is exercised; what is left is merging `next` to
# Completed Steps
- 2026-10-07: Made a backup notice a file rewritten with its size
unchanged and a new mtime in the same second as the one in the index
([issue #226](https://git.eeqj.de/sneak/vaultik/issues/226)). The
`files` table held mtime in whole seconds and the scanner compared
whole seconds, so every later snapshot kept the old content. A new
`mtime_nsec` column holds the nanoseconds within the second that
`mtime` holds, and the scanner compares the full mtime. A local index
created before the change lacks the column and is rebuilt with
`vaultik database delete` and a full backup.
- 2026-10-07: Made taking the process-wide lock atomic
([issue #227](https://git.eeqj.de/sneak/vaultik/issues/227)). The lock
read `vaultik.pid`, checked whether that PID was alive and then wrote
+2 -1
View File
@@ -36,7 +36,8 @@ Stores metadata about files in the filesystem being backed up.
**Columns:**
- `id` (TEXT PRIMARY KEY) - UUID for the file record
- `path` (TEXT NOT NULL UNIQUE) - Absolute file path
- `mtime` (INTEGER NOT NULL) - Modification time as Unix timestamp
- `mtime` (INTEGER NOT NULL) - Modification time, whole seconds since the Unix epoch
- `mtime_nsec` (INTEGER NOT NULL) - Nanoseconds within that second, 0 to 999999999
- `size` (INTEGER NOT NULL) - File size in bytes
- `mode` (INTEGER NOT NULL) - Unix file permissions and type
- `uid` (INTEGER NOT NULL) - User ID of file owner
+28 -19
View File
@@ -33,11 +33,13 @@ func (r *FileRepository) Create(ctx context.Context, tx *sql.Tx, file *File) err
}
query := `
INSERT INTO files (id, path, source_path, mtime, size, mode, uid, gid, link_target)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
INSERT INTO files
(id, path, source_path, mtime, mtime_nsec, size, mode, uid, gid, link_target)
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
ON CONFLICT(path) DO UPDATE SET
source_path = excluded.source_path,
mtime = excluded.mtime,
mtime_nsec = excluded.mtime_nsec,
size = excluded.size,
mode = excluded.mode,
uid = excluded.uid,
@@ -54,16 +56,19 @@ func (r *FileRepository) Create(ctx context.Context, tx *sql.Tx, file *File) err
if tx != nil {
LogSQL("Execute", query,
file.ID.String(), file.Path.String(), file.SourcePath.String(),
file.MTime.Unix(), file.Size, file.Mode, file.UID, file.GID,
file.MTime.Unix(), file.MTime.Nanosecond(),
file.Size, file.Mode, file.UID, file.GID,
file.LinkTarget.String())
err = tx.QueryRowContext(ctx, query,
file.ID.String(), file.Path.String(), file.SourcePath.String(),
file.MTime.Unix(), file.Size, file.Mode, file.UID, file.GID,
file.MTime.Unix(), file.MTime.Nanosecond(),
file.Size, file.Mode, file.UID, file.GID,
file.LinkTarget.String()).Scan(&idStr)
} else {
err = r.db.QueryRowWithLog(ctx, query,
file.ID.String(), file.Path.String(), file.SourcePath.String(),
file.MTime.Unix(), file.Size, file.Mode, file.UID, file.GID,
file.MTime.Unix(), file.MTime.Nanosecond(),
file.Size, file.Mode, file.UID, file.GID,
file.LinkTarget.String()).Scan(&idStr)
}
@@ -84,7 +89,7 @@ func (r *FileRepository) Create(ctx context.Context, tx *sql.Tx, file *File) err
// in the index.
func (r *FileRepository) GetByPath(ctx context.Context, path string) (*File, error) {
query := `
SELECT id, path, source_path, mtime, size, mode, uid, gid, link_target
SELECT id, path, source_path, mtime, mtime_nsec, size, mode, uid, gid, link_target
FROM files
WHERE path = ?
`
@@ -104,7 +109,7 @@ func (r *FileRepository) GetByPath(ctx context.Context, path string) (*File, err
// GetByID retrieves a file by its UUID
func (r *FileRepository) GetByID(ctx context.Context, id types.FileID) (*File, error) {
query := `
SELECT id, path, source_path, mtime, size, mode, uid, gid, link_target
SELECT id, path, source_path, mtime, mtime_nsec, size, mode, uid, gid, link_target
FROM files
WHERE id = ?
`
@@ -127,7 +132,7 @@ func (r *FileRepository) GetByPathTx(
ctx context.Context, tx *sql.Tx, path string,
) (*File, error) {
query := `
SELECT id, path, source_path, mtime, size, mode, uid, gid, link_target
SELECT id, path, source_path, mtime, mtime_nsec, size, mode, uid, gid, link_target
FROM files
WHERE path = ?
`
@@ -158,13 +163,14 @@ func (r *FileRepository) ListModifiedSince(
ctx context.Context, since time.Time,
) ([]*File, error) {
query := `
SELECT id, path, source_path, mtime, size, mode, uid, gid, link_target
SELECT id, path, source_path, mtime, mtime_nsec, size, mode, uid, gid, link_target
FROM files
WHERE mtime >= ?
WHERE (mtime, mtime_nsec) >= (?, ?)
ORDER BY path
`
rows, err := r.db.conn.QueryContext(ctx, query, since.Unix())
rows, err := r.db.conn.QueryContext(ctx, query,
since.Unix(), since.Nanosecond())
if err != nil {
return nil, fmt.Errorf("querying files: %w", err)
}
@@ -239,7 +245,7 @@ func (r *FileRepository) ListUnderPath(
// LIKE would ignore ASCII case and treat _ and % in path as wildcards.
query := `
SELECT id, path, source_path, mtime, size, mode, uid, gid, link_target
SELECT id, path, source_path, mtime, mtime_nsec, size, mode, uid, gid, link_target
FROM files
WHERE path = ? OR substr(path, 1, length(?)) = ?
ORDER BY path
@@ -324,7 +330,7 @@ func (r *FileRepository) ListIDsWithChunksNotInUploadedBlobs(
// ListAll returns all files in the database
func (r *FileRepository) ListAll(ctx context.Context) ([]*File, error) {
query := `
SELECT id, path, source_path, mtime, size, mode, uid, gid, link_target
SELECT id, path, source_path, mtime, mtime_nsec, size, mode, uid, gid, link_target
FROM files
ORDER BY path
`
@@ -365,7 +371,7 @@ func (r *FileRepository) CreateBatch(
}
// Each files row binds this many SQL variables.
const fileCols = 9
const fileCols = 10
// Batch at 100 rows to be safe with SQLite's variable limit.
const batchSize = 100
@@ -376,7 +382,7 @@ func (r *FileRepository) CreateBatch(
batch := files[i:end]
query := `INSERT INTO files
(id, path, source_path, mtime, size, mode, uid, gid, link_target)
(id, path, source_path, mtime, mtime_nsec, size, mode, uid, gid, link_target)
VALUES `
args := make([]any, 0, len(batch)*fileCols)
@@ -388,11 +394,12 @@ func (r *FileRepository) CreateBatch(
querySb325.WriteString(", ")
}
querySb325.WriteString("(?, ?, ?, ?, ?, ?, ?, ?, ?)")
querySb325.WriteString("(?, ?, ?, ?, ?, ?, ?, ?, ?, ?)")
args = append(args,
f.ID.String(), f.Path.String(), f.SourcePath.String(),
f.MTime.Unix(), f.Size, f.Mode, f.UID, f.GID,
f.MTime.Unix(), f.MTime.Nanosecond(),
f.Size, f.Mode, f.UID, f.GID,
f.LinkTarget.String())
}
@@ -401,6 +408,7 @@ func (r *FileRepository) CreateBatch(
query += ` ON CONFLICT(path) DO UPDATE SET
source_path = excluded.source_path,
mtime = excluded.mtime,
mtime_nsec = excluded.mtime_nsec,
size = excluded.size,
mode = excluded.mode,
uid = excluded.uid,
@@ -460,7 +468,7 @@ func (r *FileRepository) scanFileFrom(row fileRowScanner) (*File, error) {
var (
file File
idStr, pathStr, sourcePathStr string
mtimeUnix int64
mtimeUnix, mtimeNsec int64
linkTarget sql.NullString
)
@@ -469,6 +477,7 @@ func (r *FileRepository) scanFileFrom(row fileRowScanner) (*File, error) {
&pathStr,
&sourcePathStr,
&mtimeUnix,
&mtimeNsec,
&file.Size,
&file.Mode,
&file.UID,
@@ -487,7 +496,7 @@ func (r *FileRepository) scanFileFrom(row fileRowScanner) (*File, error) {
file.Path = types.FilePath(pathStr)
file.SourcePath = types.SourcePath(sourcePathStr)
file.MTime = time.Unix(mtimeUnix, 0).UTC()
file.MTime = time.Unix(mtimeUnix, mtimeNsec).UTC()
if linkTarget.Valid {
file.LinkTarget = types.FilePath(linkTarget.String)
}
+109
View File
@@ -252,6 +252,115 @@ func TestFileRepositorySymlink(t *testing.T) {
}
}
// An mtime after 2262 or before 1678 does not fit in int64 nanoseconds
// since the epoch, and must still come back from the database unchanged.
func TestFileRepositoryMTimeOutsideInt64NanosecondRange(t *testing.T) {
t.Parallel()
db, cleanup := setupTestDB(t)
defer cleanup()
ctx := context.Background()
repo := database.NewFileRepository(db)
mtimes := []time.Time{
time.Date(2300, time.January, 1, 0, 0, 0, 123456789, time.UTC),
time.Date(1601, time.January, 1, 0, 0, 0, 987654321, time.UTC),
}
for _, mtime := range mtimes {
created := &database.File{
Path: types.FilePath("/created-" + mtime.Format(time.RFC3339Nano)),
MTime: mtime,
}
err := repo.Create(ctx, nil, created)
if err != nil {
t.Fatalf("failed to create file: %v", err)
}
batched := &database.File{
ID: types.NewFileID(),
Path: types.FilePath("/batched-" + mtime.Format(time.RFC3339Nano)),
MTime: mtime,
}
err = repo.CreateBatch(ctx, nil, []*database.File{batched})
if err != nil {
t.Fatalf("failed to batch create file: %v", err)
}
for _, path := range []types.FilePath{created.Path, batched.Path} {
retrieved, err := repo.GetByPath(ctx, path.String())
if err != nil {
t.Fatalf("failed to get file: %v", err)
}
if !retrieved.MTime.Equal(mtime) {
t.Errorf("%s: mtime got %v, want %v",
path, retrieved.MTime, mtime)
}
}
}
}
// A file already in the index and rewritten within the same second must get
// its new nanoseconds stored, through both Create and CreateBatch.
func TestFileRepositoryUpsertMTimeInSameSecond(t *testing.T) {
t.Parallel()
db, cleanup := setupTestDB(t)
defer cleanup()
ctx := context.Background()
repo := database.NewFileRepository(db)
indexed := time.Date(2026, time.October, 7, 12, 0, 0, 100000000, time.UTC)
rewritten := indexed.Add(800 * time.Millisecond)
tests := []struct {
name string
upsert func(file *database.File) error
}{
{"Create", func(file *database.File) error {
return repo.Create(ctx, nil, file)
}},
{"CreateBatch", func(file *database.File) error {
return repo.CreateBatch(ctx, nil, []*database.File{file})
}},
}
for _, tt := range tests {
file := &database.File{
ID: types.NewFileID(),
Path: types.FilePath("/" + tt.name),
MTime: indexed,
}
err := tt.upsert(file)
if err != nil {
t.Fatalf("%s: failed to create file: %v", tt.name, err)
}
file.MTime = rewritten
err = tt.upsert(file)
if err != nil {
t.Fatalf("%s: failed to update file: %v", tt.name, err)
}
retrieved, err := repo.GetByPath(ctx, file.Path.String())
if err != nil {
t.Fatalf("%s: failed to get file: %v", tt.name, err)
}
if !retrieved.MTime.Equal(rewritten) {
t.Errorf("%s: mtime got %v, want %v",
tt.name, retrieved.MTime, rewritten)
}
}
}
func TestFileRepositoryTransaction(t *testing.T) {
t.Parallel()
@@ -606,8 +606,7 @@ func TestTimezoneHandling(t *testing.T) {
t.Skip("timezone not available")
}
// Use Truncate to remove sub-second precision since we store as Unix timestamps
nyTime := time.Now().In(loc).Truncate(time.Second)
nyTime := time.Now().In(loc)
file := &File{
Path: "/timezone-test.txt",
MTime: nyTime,
+2 -1
View File
@@ -6,7 +6,8 @@ CREATE TABLE IF NOT EXISTS files (
id TEXT PRIMARY KEY, -- UUID
path TEXT NOT NULL UNIQUE,
source_path TEXT NOT NULL DEFAULT '', -- The source directory this file came from (for restore path stripping)
mtime INTEGER NOT NULL,
mtime INTEGER NOT NULL, -- whole seconds since the Unix epoch
mtime_nsec INTEGER NOT NULL, -- nanoseconds within that second, 0 to 999999999
size INTEGER NOT NULL,
mode INTEGER NOT NULL,
uid INTEGER NOT NULL,
+1 -1
View File
@@ -1221,7 +1221,7 @@ func (s *Scanner) checkFileInMemory(
// Check if file has changed
if existingFile.Size != file.Size ||
existingFile.MTime.Unix() != file.MTime.Unix() ||
!existingFile.MTime.Equal(file.MTime) ||
existingFile.Mode != file.Mode ||
existingFile.UID != file.UID ||
existingFile.GID != file.GID {
@@ -0,0 +1,73 @@
package vaultik_test
import (
"bytes"
"context"
"path/filepath"
"testing"
"time"
"github.com/spf13/afero"
"github.com/stretchr/testify/require"
"sneak.berlin/go/vaultik/internal/database"
"sneak.berlin/go/vaultik/internal/log"
"sneak.berlin/go/vaultik/internal/storage"
"sneak.berlin/go/vaultik/internal/vaultik"
)
// A file rewritten with its size unchanged and a new mtime in the same
// second as the mtime the index holds must still be backed up. See
// https://git.eeqj.de/sneak/vaultik/issues/226.
//
//nolint:paralleltest // installs the global logger via log.Initialize
func TestBackupOfSameSecondRewriteRestoresNewContent(t *testing.T) {
log.Initialize(log.Config{})
fs := afero.NewOsFs()
tempDir := t.TempDir()
dataDir := filepath.Join(tempDir, "src")
storeDir := filepath.Join(tempDir, "remote")
restoreDir := filepath.Join(tempDir, "restored")
dbPath := filepath.Join(tempDir, "index.sqlite")
rewrittenPath := filepath.Join(dataDir, "small.txt")
ctx := context.Background()
files := writeFaultSourceTree(t, fs, dataDir)
cfg := changedFileConfig(dataDir, dbPath)
firstMTime := time.Date(2026, time.January, 2, 3, 4, 5, 0, time.UTC).
Add(100 * time.Millisecond)
secondMTime := firstMTime.Add(800 * time.Millisecond)
require.NoError(t, fs.Chtimes(rewrittenPath, firstMTime, firstMTime))
store, err := storage.NewFileStorer(storeDir)
require.NoError(t, err)
db, err := database.New(ctx, dbPath)
require.NoError(t, err)
repos := database.NewRepositories(db)
v := newBackupVaultik(ctx, cfg, store, repos, db, fs)
require.NoError(t, backUp(v, "first"))
// Upper-casing ASCII text keeps its size.
files[rewrittenPath] = bytes.ToUpper(files[rewrittenPath])
require.NoError(t, afero.WriteFile(fs, rewrittenPath, files[rewrittenPath], 0o644))
require.NoError(t, fs.Chtimes(rewrittenPath, secondMTime, secondMTime))
require.NoError(t, backUp(v, "second"))
id := localSnapshotID(ctx, t, repos, "second")
require.NoError(t, db.Close())
reader := newReaderVaultik(ctx, cfg, store, nil, fs)
require.NoError(t, reader.Restore(&vaultik.RestoreOptions{
SnapshotID: id,
TargetDir: restoreDir,
Verify: true,
}))
assertRestoredTree(t, fs, restoreDir, files)
}