Delete .tmp- leftovers of a killed command when the lock is next taken (closes #75)
check / check (push) Failing after 3s
check / check (push) Failing after 3s
A command killed part-way could leave a temporary file or directory of secret.WriteFileAtomic or secret.TempDirFor, encrypted keys included, for good. LockStateDir now empties the lock file once it holds the lock and writes "finished" there just before releasing it. A holder that does not find that deletes such leftovers from the state directory, each vault, each secret and each version, the only places those helpers make them, matching names that start with "." and hold ".tmp-". After a command that finished nothing is searched, so the added time does not grow with the number of secrets and versions. A test shows that `unlocker remove` removes an unlocker directory with no metadata file. Model: opus-5-5
This commit is contained in:
+95
-2
@@ -1,6 +1,7 @@
|
||||
package vault
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
@@ -14,6 +15,11 @@ import (
|
||||
// lockFileName is the file in the state directory that LockStateDir locks.
|
||||
const lockFileName = "lock"
|
||||
|
||||
// finishedMark is what the lock file holds once the command that last held
|
||||
// the lock has released it. A command killed while holding it leaves the
|
||||
// file empty.
|
||||
const finishedMark = "finished\n"
|
||||
|
||||
// memFsLock stands in for the lock file on the in-memory filesystem, which
|
||||
// has no file locks. Every in-memory filesystem in the process shares it.
|
||||
//
|
||||
@@ -25,6 +31,12 @@ var memFsLock sync.Mutex
|
||||
// it. While one command holds it, the next one waits here. Reads take no
|
||||
// lock: each file or directory a command changes is replaced in a single
|
||||
// rename, so a reader finds it as it was before or after, never half-made.
|
||||
// Once it holds the lock, it empties the lock file, and the function it
|
||||
// returns writes finishedMark there just before releasing the lock, so a
|
||||
// command killed while holding the lock leaves the mark missing. Finding it
|
||||
// missing, LockStateDir first deletes the temporary files and directories
|
||||
// such a command may have left, since no command still using them can be
|
||||
// running. After a command that finished, it searches nothing.
|
||||
//
|
||||
// On the real filesystem the lock is flock(2) on the file "lock" in
|
||||
// stateDir, which the kernel releases when the process dies, so a killed
|
||||
@@ -32,16 +44,97 @@ var memFsLock sync.Mutex
|
||||
// use has no file locks, so a process-wide mutex stands in for flock there.
|
||||
// Any other filesystem is refused rather than left unlocked.
|
||||
func LockStateDir(fs afero.Fs, stateDir string) (func(), error) {
|
||||
var release func()
|
||||
|
||||
switch fs.(type) {
|
||||
case *afero.OsFs:
|
||||
return flockStateDir(stateDir)
|
||||
var err error
|
||||
|
||||
release, err = flockStateDir(stateDir)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
case *afero.MemMapFs:
|
||||
memFsLock.Lock()
|
||||
|
||||
return memFsLock.Unlock, nil
|
||||
release = memFsLock.Unlock
|
||||
default:
|
||||
return nil, fmt.Errorf("%w %T", ErrNoLockForFilesystem, fs)
|
||||
}
|
||||
|
||||
// The lock file is written in place, never replaced: a command waiting
|
||||
// for flock on the old file would then take a lock nobody else checks.
|
||||
lockPath := filepath.Join(stateDir, lockFileName)
|
||||
|
||||
mark, err := afero.ReadFile(fs, lockPath)
|
||||
if err != nil || string(mark) != finishedMark {
|
||||
removeLeftovers(fs, stateDir)
|
||||
}
|
||||
|
||||
err = afero.WriteFile(fs, lockPath, nil, secret.FilePerms)
|
||||
if err != nil {
|
||||
release()
|
||||
|
||||
return nil, fmt.Errorf("failed to empty lock file %s: %w", lockPath, err)
|
||||
}
|
||||
|
||||
return func() {
|
||||
// If this fails, the next command searches when it need not.
|
||||
_ = afero.WriteFile(fs, lockPath, []byte(finishedMark), secret.FilePerms)
|
||||
|
||||
release()
|
||||
}, nil
|
||||
}
|
||||
|
||||
// removeLeftovers deletes the temporary files and directories that commands
|
||||
// killed part-way left in each directory where secret.WriteFileAtomic and
|
||||
// secret.TempDirFor make them: the state directory, each vault, each secret
|
||||
// and each version. Unlocker directories are written whole by
|
||||
// secret.WriteDir and never changed after, so they hold none. A failure is
|
||||
// only warned about, and the command goes on.
|
||||
func removeLeftovers(fs afero.Fs, stateDir string) {
|
||||
dirs := []string{stateDir}
|
||||
|
||||
for _, vaultDir := range subdirs(fs, filepath.Join(stateDir, "vaults.d")) {
|
||||
dirs = append(dirs, vaultDir)
|
||||
|
||||
for _, secretDir := range subdirs(fs, filepath.Join(vaultDir, "secrets.d")) {
|
||||
dirs = append(dirs, secretDir)
|
||||
dirs = append(dirs, subdirs(fs, filepath.Join(secretDir, "versions"))...)
|
||||
}
|
||||
}
|
||||
|
||||
for _, dir := range dirs {
|
||||
err := secret.RemoveLeftovers(fs, dir)
|
||||
if err != nil {
|
||||
secret.Warn("Failed to remove what an interrupted command left",
|
||||
"error", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// subdirs returns the directories in dir: none if dir does not exist, and
|
||||
// none, with a warning, if it cannot be read.
|
||||
func subdirs(fs afero.Fs, dir string) []string {
|
||||
entries, err := afero.ReadDir(fs, dir)
|
||||
if err != nil {
|
||||
if !errors.Is(err, os.ErrNotExist) {
|
||||
secret.Warn("Failed to look for what an interrupted command left",
|
||||
"directory", dir, "error", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
var dirs []string
|
||||
|
||||
for _, entry := range entries {
|
||||
if entry.IsDir() {
|
||||
dirs = append(dirs, filepath.Join(dir, entry.Name()))
|
||||
}
|
||||
}
|
||||
|
||||
return dirs
|
||||
}
|
||||
|
||||
// flockStateDir takes flock(2) on the lock file in stateDir, creating the
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
package vault_test
|
||||
|
||||
import (
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.eeqj.de/sneak/secret/internal/secret"
|
||||
"git.eeqj.de/sneak/secret/internal/vault"
|
||||
"github.com/spf13/afero"
|
||||
"github.com/stretchr/testify/assert"
|
||||
@@ -121,6 +123,51 @@ func TestLockStateDirFreeAfterPanic(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestLockStateDirRemovesLeftoversOnlyAfterKill checks that taking the lock
|
||||
// deletes a temporary directory a killed command left only when the last
|
||||
// holder of the lock did not release it. A holder killed while it holds the
|
||||
// lock leaves the lock file as it is at that moment.
|
||||
func TestLockStateDirRemovesLeftoversOnlyAfterKill(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
for _, lfs := range lockFilesystems(t) {
|
||||
t.Run(lfs.name, func(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
lockFile := filepath.Join(lfs.stateDir, "lock")
|
||||
leftover := filepath.Join(lfs.stateDir, ".tmp-1")
|
||||
|
||||
release, err := vault.LockStateDir(lfs.fs, lfs.stateDir)
|
||||
require.NoError(t, err)
|
||||
|
||||
whileHeld, err := afero.ReadFile(lfs.fs, lockFile)
|
||||
require.NoError(t, err)
|
||||
release()
|
||||
|
||||
require.NoError(t, lfs.fs.MkdirAll(leftover, secret.DirPerms))
|
||||
|
||||
release, err = vault.LockStateDir(lfs.fs, lfs.stateDir)
|
||||
require.NoError(t, err)
|
||||
release()
|
||||
|
||||
exists, err := afero.DirExists(lfs.fs, leftover)
|
||||
require.NoError(t, err)
|
||||
assert.True(t, exists, "searched after a holder that finished")
|
||||
|
||||
require.NoError(t, afero.WriteFile(lfs.fs, lockFile, whileHeld,
|
||||
secret.FilePerms))
|
||||
|
||||
release, err = vault.LockStateDir(lfs.fs, lfs.stateDir)
|
||||
require.NoError(t, err)
|
||||
release()
|
||||
|
||||
exists, err = afero.DirExists(lfs.fs, leftover)
|
||||
require.NoError(t, err)
|
||||
assert.False(t, exists, "not searched after a holder that was killed")
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestLockStateDirRefusesOtherFilesystems checks that a filesystem with no
|
||||
// lock implementation is refused instead of being used unlocked.
|
||||
func TestLockStateDirRefusesOtherFilesystems(t *testing.T) {
|
||||
|
||||
Reference in New Issue
Block a user