Stop scan cleanly on SIGINT or SIGTERM (closes #5)
check / check (push) Failing after 2s

A first SIGINT or SIGTERM cancels the scan. It commits the hashed
records still in its batch, with a context that is not cancelled for
that one write, and starts no other write or deletion; deletions need
a complete walk, so records under paths an interrupted walk never
reached are kept. A batch whose commit failed is now kept for that
final commit instead of dropped. The progress display is finished (a
bar stopped short is no longer filled up), `scan: interrupted after N
files` goes to stderr, and the exit code is 1. A second signal ends the
process at once. A SIGINT inherited as ignored stays ignored.

Model: opus-5-5
This commit is contained in:
2026-10-04 06:56:41 +00:00
committed by sneak
parent 33cf3dd29a
commit 1ed3df06f2
8 changed files with 446 additions and 62 deletions
+247 -13
View File
@@ -4,11 +4,16 @@ import (
"context"
"database/sql"
"errors"
"fmt"
"os"
"os/signal"
"path/filepath"
"slices"
"strconv"
"strings"
"sync"
"sync/atomic"
"syscall"
"testing"
"time"
)
@@ -150,9 +155,8 @@ func assertRecordsIntact(t *testing.T, db *sql.DB, before []string) {
// Every one of those records would look vanished to the update phase.
// The guard is what stops the scan there, and this test is what
// notices if it stops doing so: deleting the guard, or making it
// unreachable, makes the scan carry its truncated view into a later
// phase and fail there instead, with a wrapped error rather than the
// bare cancellation.
// unreachable, makes the scan carry its truncated view into the update
// phase, which counts every record the walk never reached for removal.
//
//nolint:paralleltest // counts goroutines: must not run beside others
func TestSyncScanCancelledMidWalkKeepsRecords(t *testing.T) {
@@ -182,10 +186,10 @@ func TestSyncScanCancelledMidWalkKeepsRecords(t *testing.T) {
// assertWalkGuardAborted checks that the scan stopped at the post-walk
// guard: with a census that is neither empty (the walk really ran)
// nor complete (it really was cut short), and with the guard's own
// bare cancellation as the error. A wrapped error means the partial
// census was carried past the guard into the hash or update phase,
// which is the failure this test exists to catch.
// nor complete (it really was cut short), and with no record counted
// for removal. A removal count means the partial census was carried
// past the guard into the update phase, which is the failure this test
// exists to catch.
func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
t.Helper()
@@ -194,12 +198,6 @@ func assertWalkGuardAborted(t *testing.T, st scanStats, err error) {
err, context.Canceled)
}
if errors.Unwrap(err) != nil {
t.Errorf("syncScan reported %q, want the guard's bare "+
"cancellation: a wrapped error means the truncated census "+
"reached a later phase", err)
}
if st.unchanged == 0 {
t.Fatalf("stats = %+v: the census is empty, so the walk never "+
"ran and the guard was reached for the wrong reason", st)
@@ -260,6 +258,242 @@ func TestSyncScanCancelledBeforeLoadIndex(t *testing.T) {
assertRecordsIntact(t, db, before)
}
// hashCancelAtDone is the consultation on which the mid-hash test's
// context cancels itself. The walk of buildWalkCancelTree spends about
// one per file and three per directory, and the hash phase then one per
// file hashed, so this lands about half way through the hash phase.
const hashCancelAtDone = walkCancelFiles + 3*walkCancelDirs +
walkCancelFiles/2
// TestSyncScanCancelledMidHashKeepsHashedRecords cancels a first scan
// part-way through its hash phase. The fixture holds fewer files than a
// batch, so every file hashed is still waiting to be committed: the scan
// must commit them all before it returns, and the next scan must hash
// only the rest.
func TestSyncScanCancelledMidHashKeepsHashedRecords(t *testing.T) {
t.Parallel()
dir := buildWalkCancelTree(t)
db := openTestDB(t)
st, err := syncScan(newWalkClock(hashCancelAtDone), db,
[]string{dir}, walkCancelWorkers, false)
if !errors.Is(err, context.Canceled) {
t.Fatalf("syncScan cancelled mid-hash = %v, want %v",
err, context.Canceled)
}
if st.walked != walkCancelFiles || st.added == 0 ||
st.added >= walkCancelFiles {
t.Fatalf("stats = %+v: want the walk complete and the hash phase "+
"cut short", st)
}
if got := len(dbRecords(t, db)); got != st.added {
t.Errorf("%d records after the cancelled scan, want the %d it hashed",
got, st.added)
}
hashed := st.added
st = syncTree(t, db, dir)
if st.added != walkCancelFiles-hashed || st.unchanged != hashed {
t.Errorf("next scan stats = %+v, want %d added %d unchanged",
st, walkCancelFiles-hashed, hashed)
}
}
// storedPaths opens the database at path as report does, which fails
// unless it is a valid database, and returns its records' paths.
func storedPaths(t *testing.T, path string) []string {
t.Helper()
db, err := openReportDatabase(t.Context(), path)
if err != nil {
t.Fatal(err)
}
defer func() { _ = db.Close() }()
return recordPaths(dbRecords(t, db))
}
// TestRunScanInterrupted calls the scan entrypoint with a context that
// is already cancelled, as when a signal arrives at once. It must return
// errInterrupted promptly with its one line on stderr, leave the
// database valid and as it was, and leave nothing in the way of the
// next scan, which must bring the database up to date.
func TestRunScanInterrupted(t *testing.T) {
path := testDBPath(t)
t.Setenv(databaseEnv, path)
stderr := captureStderr(t)
dir := buildSmokeTree(t)
err := runScan(t.Context(), []string{dir}, walkCancelWorkers, false)
if err != nil {
t.Fatal(err)
}
before := storedPaths(t, path)
// A vanished file and a new one: the interrupted scan records
// neither.
gone := filepath.Join(dir, "a", "unique.bin")
err = os.Remove(gone)
if err != nil {
t.Fatal(err)
}
added := writeFile(t, dir, "a/new.bin", pattern(50, 10))
shown := len(stderr())
done := make(chan struct{})
go func() {
defer close(done)
err = runScan(cancelledContext(t), []string{dir}, walkCancelWorkers,
false)
}()
awaitReturn(t, done, "runScan")
if !errors.Is(err, errInterrupted) {
t.Fatalf("runScan on a cancelled context = %v, want %v",
err, errInterrupted)
}
want := "scan: interrupted after 0 files\n"
if got := stderr()[shown:]; got != want {
t.Errorf("stderr = %q, want %q", got, want)
}
assertNoSidecars(t, path)
if got := storedPaths(t, path); !slices.Equal(got, before) {
t.Errorf("records = %q after the interrupted scan, want %q",
got, before)
}
err = runScan(t.Context(), []string{dir}, walkCancelWorkers, false)
if err != nil {
t.Fatal(err)
}
got := storedPaths(t, path)
if slices.Contains(got, gone) || !slices.Contains(got, added) {
t.Errorf("records = %q after the next scan, want %q gone and %q "+
"added", got, gone, added)
}
}
// TestRunScanInterruptedMidHash interrupts the scan entrypoint part-way
// through its hash phase, after the database is open. It must return
// errInterrupted, release the lock, end stderr with its line counting
// every file the walk reached, close the database out of WAL mode, and
// keep the records it hashed.
func TestRunScanInterruptedMidHash(t *testing.T) {
path := testDBPath(t)
t.Setenv(databaseEnv, path)
stderr := captureStderr(t)
dir := buildWalkCancelTree(t)
err := runScan(newWalkClock(hashCancelAtDone), []string{dir},
walkCancelWorkers, false)
if !errors.Is(err, errInterrupted) {
t.Fatalf("runScan interrupted mid-hash = %v, want %v",
err, errInterrupted)
}
holdScanLock(t, path)
want := fmt.Sprintf("scan: interrupted after %d files\n", walkCancelFiles)
if got := stderr(); !strings.HasSuffix(got, want) {
t.Errorf("stderr = %q, want it to end with %q", got, want)
}
assertNoSidecars(t, path)
db, err := openReportDatabase(t.Context(), path)
if err != nil {
t.Fatal(err)
}
defer func() { _ = db.Close() }()
// A plain close also removes the sidecars, but leaves WAL mode on.
var mode string
err = db.QueryRowContext(t.Context(), "PRAGMA journal_mode").Scan(&mode)
if err != nil {
t.Fatal(err)
}
if mode != "delete" {
t.Errorf("journal mode = %q after the interrupted scan, want %q",
mode, "delete")
}
kept := len(dbRecords(t, db))
if kept == 0 || kept >= walkCancelFiles {
t.Errorf("%d records after the interrupted scan, want those it "+
"hashed: some but not all of the %d files", kept, walkCancelFiles)
}
}
// TestInterruptContextCatchesSIGTERM sends SIGTERM to the test process
// while the scan's handler is installed, and checks that it cancels the
// scan's context.
//
//nolint:paralleltest // signals the whole process: must not run beside a scan
func TestInterruptContextCatchesSIGTERM(t *testing.T) {
// Caught here as well, so that a handler that misses SIGTERM fails
// this test instead of ending the test process.
caught := make(chan os.Signal, 1)
signal.Notify(caught, syscall.SIGTERM)
defer signal.Stop(caught)
ctx, stop := interruptContext(t.Context())
defer stop()
err := syscall.Kill(os.Getpid(), syscall.SIGTERM)
if err != nil {
t.Fatal(err)
}
select {
case <-ctx.Done():
case <-time.After(poolUnwind):
t.Fatal("SIGTERM did not cancel the scan's context")
}
}
// TestCommitFullBatchKeepsFailedBatch checks that a full batch whose
// commit fails, as it does once the scan is interrupted, stays in the
// batch, so that syncScan's final commit saves it.
func TestCommitFullBatchKeepsFailedBatch(t *testing.T) {
t.Parallel()
s := &scanState{db: openTestDB(t)}
for i := range updateBatchSize {
s.batch = append(s.batch, scanRec{path: "/f" + strconv.Itoa(i)})
}
err := s.commitFullBatch(cancelledContext(t))
if !errors.Is(err, context.Canceled) {
t.Fatalf("commitFullBatch on a cancelled context = %v, want %v",
err, context.Canceled)
}
if len(s.batch) != updateBatchSize {
t.Errorf("batch holds %d records after the failed commit, want %d",
len(s.batch), updateBatchSize)
}
}
// drainClosed counts the values received from ch until it closes,
// failing the test if it does not close within poolUnwind. A pool that
// ignored its cancellation leaves its channel open with its goroutines