A first SIGINT or SIGTERM cancels the scan. It commits the hashed records still in its batch, with a context that is not cancelled for that one write, and starts no other write or deletion; deletions need a complete walk, so records under paths an interrupted walk never reached are kept. A batch whose commit failed is now kept for that final commit instead of dropped. The progress display is finished (a bar stopped short is no longer filled up), `scan: interrupted after N files` goes to stderr, and the exit code is 1. A second signal ends the process at once. A SIGINT inherited as ignored stays ignored. Model: opus-5-5
This commit was merged in pull request #79.
This commit is contained in:
@@ -11,6 +11,7 @@ import (
|
||||
"io"
|
||||
"io/fs"
|
||||
"os"
|
||||
"os/signal"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strings"
|
||||
@@ -57,6 +58,10 @@ const sampleWindow = 1024 * 1024
|
||||
// and hash worker pools.
|
||||
const workQueueDepth = 1024
|
||||
|
||||
// errInterrupted reports a scan stopped by SIGINT or SIGTERM. runScan
|
||||
// has already printed its line, so run prints nothing more.
|
||||
var errInterrupted = errors.New("scan interrupted")
|
||||
|
||||
// fileRec carries one statted file between the scan phases. dev and
|
||||
// ino identify the underlying inode so hard-linked paths can share
|
||||
// one read; both are zero when the platform exposes no inode.
|
||||
@@ -90,8 +95,10 @@ type fileMeta struct {
|
||||
// scan fails before it walks the filesystem or opens the database.
|
||||
// Errors are returned rather than exiting, so that the deferred close —
|
||||
// which takes the database out of WAL mode — always runs, and the lock
|
||||
// is released after it. Cancelling ctx unwinds the worker pools and
|
||||
// aborts the scan with the context's error.
|
||||
// is released after it. When ctx is cancelled, as by the SIGINT or
|
||||
// SIGTERM that interruptContext catches, the scan keeps what it has
|
||||
// hashed (see syncScan), prints how many files its walk reached, and
|
||||
// returns errInterrupted.
|
||||
func runScan(ctx context.Context, roots []string, workers int,
|
||||
oneFS bool,
|
||||
) error {
|
||||
@@ -114,6 +121,12 @@ func runScan(ctx context.Context, roots []string, workers int,
|
||||
defer func() { _ = lock.Close() }()
|
||||
|
||||
db, err := openScanDatabase(ctx, dbPath)
|
||||
if err != nil && ctx.Err() != nil {
|
||||
// Interrupted while opening; SQLite may report that with an
|
||||
// error of its own rather than the context's.
|
||||
return interrupted(0)
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -121,6 +134,10 @@ func runScan(ctx context.Context, roots []string, workers int,
|
||||
defer closeScanDatabase(ctx, db, dbPath)
|
||||
|
||||
st, err := syncScan(ctx, db, roots, workers, oneFS)
|
||||
if errors.Is(err, context.Canceled) {
|
||||
return interrupted(st.walked)
|
||||
}
|
||||
|
||||
if err != nil {
|
||||
return fmt.Errorf("update database %s: %w", dbPath, err)
|
||||
}
|
||||
@@ -134,6 +151,34 @@ func runScan(ctx context.Context, roots []string, workers int,
|
||||
return nil
|
||||
}
|
||||
|
||||
// interruptContext returns a copy of ctx that the first SIGINT or
|
||||
// SIGTERM cancels; the scan command runs the scan under it. stop
|
||||
// releases the signals.
|
||||
func interruptContext(ctx context.Context) (context.Context, func()) {
|
||||
// A SIGINT ignored from the start, as by a script's background job,
|
||||
// stays ignored.
|
||||
signals := []os.Signal{syscall.SIGTERM}
|
||||
if !signal.Ignored(syscall.SIGINT) {
|
||||
signals = append(signals, syscall.SIGINT)
|
||||
}
|
||||
|
||||
ctx, stop := signal.NotifyContext(ctx, signals...)
|
||||
|
||||
// Stopping restores the default handling, so a second signal ends
|
||||
// the process at once.
|
||||
context.AfterFunc(ctx, stop)
|
||||
|
||||
return ctx, stop
|
||||
}
|
||||
|
||||
// interrupted prints the line for a scan stopped by a signal after its
|
||||
// walk reached walked files, and returns errInterrupted.
|
||||
func interrupted(walked int) error {
|
||||
fmt.Fprintf(os.Stderr, "scan: interrupted after %d files\n", walked)
|
||||
|
||||
return errInterrupted
|
||||
}
|
||||
|
||||
// resolveRoots converts each PATH operand to an absolute, lexically
|
||||
// cleaned path (symlinks are not resolved) and verifies that it
|
||||
// exists. Database records are keyed by absolute path, so scan results
|
||||
@@ -187,8 +232,9 @@ func pruneRoots(roots []string) []string {
|
||||
}
|
||||
|
||||
// scanStats summarizes one scan's database synchronization for the
|
||||
// final stderr summary.
|
||||
// final stderr summary, or for the line an interrupted scan prints.
|
||||
type scanStats struct {
|
||||
walked int // files the walk reached
|
||||
added int
|
||||
updated int
|
||||
removed int
|
||||
@@ -210,7 +256,30 @@ type scanState struct {
|
||||
st scanStats
|
||||
}
|
||||
|
||||
// syncScan synchronizes the database with the filesystem under roots
|
||||
// syncScan synchronizes the database with the filesystem under roots;
|
||||
// see runPhases. When ctx is cancelled, as by an interrupt, it commits
|
||||
// the hashed records still waiting in the batch, starts no other write
|
||||
// or deletion, and returns the cancellation.
|
||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||
workers int, oneFS bool,
|
||||
) (scanStats, error) {
|
||||
s := &scanState{db: db}
|
||||
|
||||
err := s.runPhases(ctx, roots, workers, oneFS)
|
||||
if err == nil || ctx.Err() == nil {
|
||||
return s.st, err
|
||||
}
|
||||
|
||||
// The one write made after the cancellation, so it cannot use ctx.
|
||||
err = applyChanges(context.WithoutCancel(ctx), db, s.batch, nil, nil)
|
||||
if err != nil {
|
||||
return s.st, err
|
||||
}
|
||||
|
||||
return s.st, ctx.Err()
|
||||
}
|
||||
|
||||
// runPhases synchronizes the database with the filesystem under roots
|
||||
// in four sequential phases: walk (enumerate and stat every file,
|
||||
// building a complete size census), hash (read only the new or
|
||||
// changed — or previously unhashed — files whose size at least one
|
||||
@@ -223,48 +292,41 @@ type scanState struct {
|
||||
// their content hash. Operands the walk cannot start from are dropped
|
||||
// first, so the records beneath them count as outside the roots unless
|
||||
// they lie under another root.
|
||||
func syncScan(ctx context.Context, db *sql.DB, roots []string,
|
||||
func (s *scanState) runPhases(ctx context.Context, roots []string,
|
||||
workers int, oneFS bool,
|
||||
) (scanStats, error) {
|
||||
s := &scanState{db: db}
|
||||
|
||||
) error {
|
||||
// Types are checked before pruning so that an operand under a
|
||||
// dropped one is still scanned, not dropped as lying under it.
|
||||
roots = pruneRoots(s.walkableRoots(roots))
|
||||
|
||||
err := s.loadIndex(ctx, roots)
|
||||
if err != nil {
|
||||
return s.st, err
|
||||
return err
|
||||
}
|
||||
|
||||
changed, unhashed := s.walkPhase(startWalk(ctx, roots, oneFS, workers))
|
||||
|
||||
// A cancelled walk stops early, so its size census covers only part
|
||||
// of the roots, and every file it never reached looks vanished to
|
||||
// the update phase. Defence in depth rather than the only barrier:
|
||||
// that phase would today fail on its first BeginTx with the same
|
||||
// cancelled context before deleting anything. But it is the barrier
|
||||
// that survives a later decision to let an interrupted scan commit
|
||||
// what it has, and it turns a confusing failure deep in the update
|
||||
// phase into a clean abort at the phase boundary.
|
||||
// of the roots, and every file it never reached would look vanished
|
||||
// to the update phase. Stop before anything is written or deleted.
|
||||
err = ctx.Err()
|
||||
if err != nil {
|
||||
return s.st, err
|
||||
return err
|
||||
}
|
||||
|
||||
s.partition(changed, unhashed)
|
||||
|
||||
err = s.hashPhase(ctx, workers)
|
||||
if err != nil {
|
||||
return s.st, err
|
||||
return err
|
||||
}
|
||||
|
||||
err = s.updatePhase(ctx)
|
||||
if err != nil {
|
||||
return s.st, err
|
||||
return err
|
||||
}
|
||||
|
||||
return s.st, s.contentPhase(ctx, workers)
|
||||
return s.contentPhase(ctx, workers)
|
||||
}
|
||||
|
||||
// walkableRoots returns the operands the walk can start from: regular
|
||||
@@ -349,6 +411,7 @@ func (s *scanState) walkPhase(
|
||||
}
|
||||
|
||||
s.sizes = append(s.sizes, ev.rec.size)
|
||||
s.st.walked++
|
||||
|
||||
prog.increment()
|
||||
|
||||
@@ -552,16 +615,22 @@ func (s *scanState) recordRun(ctx context.Context, r hashResult) error {
|
||||
}
|
||||
|
||||
// commitFullBatch commits the running batch once it holds
|
||||
// updateBatchSize records.
|
||||
// updateBatchSize records. A batch that fails to commit is kept: the
|
||||
// commit fails when the scan is interrupted, and syncScan then commits
|
||||
// the batch itself.
|
||||
func (s *scanState) commitFullBatch(ctx context.Context) error {
|
||||
if len(s.batch) < updateBatchSize {
|
||||
return nil
|
||||
}
|
||||
|
||||
err := applyBatch(ctx, s.db, s.batch, nil, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
s.batch = s.batch[:0]
|
||||
|
||||
return err
|
||||
return nil
|
||||
}
|
||||
|
||||
// updatePhase writes the scan's tail under one progress display: the
|
||||
@@ -1004,6 +1073,12 @@ func walkOneDir(ctx context.Context, job dirJob, oneFS bool,
|
||||
var subs []dirJob
|
||||
|
||||
for _, e := range entries {
|
||||
// A cancelled scan wants nothing more from this directory: stop
|
||||
// rather than lstat the rest of a large one.
|
||||
if ctx.Err() != nil {
|
||||
return nil
|
||||
}
|
||||
|
||||
p := filepath.Join(job.path, e.Name())
|
||||
|
||||
if e.IsDir() {
|
||||
|
||||
Reference in New Issue
Block a user