Files
vaultik/internal/vaultik/blobcache.go
T
sneak 075c5e9733
check / check (pull_request) Successful in 1m23s
Validate blob hashes, offsets and lengths from the destination (closes #155)
A blob hash read back from the downloaded snapshot database or the store
listing was trusted unchecked. A hostile remote could set a hash such as
"aa/../../etc" and make a decrypted blob be written outside the cache
directory, or feed a short or negative value that panicked a command.

blobDiskCache.path now refuses any key containing a path separator, and
ReadAt rejects a negative offset or length, bounding with
length > size-offset so a sum cannot overflow past the check. A new
isBlobHash helper (a plain function, not a method, since the packer stores
temp-placeholder-{uuid} as a hash) gates FetchBlob, shallow and deep
verify; the blobs/ and metadata/ listings skip a non-conforming name with
a warning; and short-hash prefixes in log and error text go through a
shortHash helper that cannot panic. verify's chunk reader rejects a
negative blob_chunks length and streams the chunk rather than allocating a
database-supplied size.

restore.go and internal/database are left untouched to avoid colliding
with the in-flight issue #156 work; the cache-path and FetchBlob guards
already stop the unsafe write and fetch.

Model: opus-4-8
2026-09-22 13:00:41 +00:00

454 lines
10 KiB
Go

package vaultik
import (
"errors"
"fmt"
"io"
"os"
"path/filepath"
"strings"
"sync"
)
// Sentinel errors for blob cache lookups.
var (
errCacheKeyMissing = errors.New("key not in cache")
errCacheReadBeyondBlob = errors.New("read beyond blob size")
errCacheKeyHasSeparator = errors.New(
"cache key contains a path separator")
errCacheNegativeRead = errors.New("negative offset or length")
)
// blobCacheFileMode is the permission mode for cached blob files.
const blobCacheFileMode = 0o600
// blobDiskCacheEntry tracks a cached blob on disk.
type blobDiskCacheEntry struct {
key string
size int64
prev *blobDiskCacheEntry
next *blobDiskCacheEntry
}
// blobDiskCache stores blobs on disk keyed by hash. It exposes ReadAt
// for slice reads (the restore path uses this so chunk extraction
// never reads a whole blob into memory) plus Get/Put for whole-blob
// access.
//
// Eviction policy is caller-controlled. The cache keeps an LRU list
// internally and will fall back to LRU eviction if curBytes exceeds
// maxBytes. Restore passes math.MaxInt64 as maxBytes and drives
// eviction itself via Delete() through restoreSweeper, which deletes
// each blob the moment every file that references its chunks has been
// written. LRU never fires under that configuration; it is kept as a
// safety net for callers that don't manage eviction themselves.
//
// Get/ReadAt/peak-Len counters are debugging instrumentation used by
// tests to assert that the restore code path uses ReadAt rather than
// Get and to bound peak disk-cache occupancy.
type blobDiskCache struct {
mu sync.Mutex
dir string
maxBytes int64
curBytes int64
items map[string]*blobDiskCacheEntry
head *blobDiskCacheEntry // most recent
tail *blobDiskCacheEntry // least recent
// Instrumentation. Mutated under mu; readable via the methods below.
getCalls int
readAtCalls int
peakLen int
}
// newBlobDiskCache creates a new disk-based blob cache with the given max size.
func newBlobDiskCache(maxBytes int64) (*blobDiskCache, error) {
dir, err := os.MkdirTemp("", "vaultik-blobcache-*")
if err != nil {
return nil, fmt.Errorf("creating blob cache dir: %w", err)
}
return &blobDiskCache{
dir: dir,
maxBytes: maxBytes,
items: make(map[string]*blobDiskCacheEntry),
}, nil
}
// Put writes blob data to disk cache. Entries larger than maxBytes are
// silently skipped.
func (c *blobDiskCache) Put(key string, data []byte) error {
p, err := c.path(key)
if err != nil {
return err
}
entrySize := int64(len(data))
c.mu.Lock()
defer c.mu.Unlock()
if entrySize > c.maxBytes {
return nil
}
// Remove old entry if updating
if e, ok := c.items[key]; ok {
c.unlink(e)
c.curBytes -= e.size
_ = os.Remove(p)
delete(c.items, key)
}
err = os.WriteFile(p, data, blobCacheFileMode)
if err != nil {
return fmt.Errorf("writing blob to cache: %w", err)
}
e := &blobDiskCacheEntry{key: key, size: entrySize}
c.pushFront(e)
c.items[key] = e
c.curBytes += entrySize
for c.curBytes > c.maxBytes && c.tail != nil {
c.evictLRU()
}
if n := len(c.items); n > c.peakLen {
c.peakLen = n
}
return nil
}
// PutFromReader streams r into the cache file for key, returning the
// total number of bytes written. Unlike Put, the data never has to
// reside fully in memory at any point — io.Copy uses an internal
// 32 KiB buffer. Used by restore to land a freshly decrypted blob on
// disk without buffering its entire plaintext (which may be tens of GB)
// in RAM.
func (c *blobDiskCache) PutFromReader(key string, r io.Reader) (int64, error) {
p, err := c.path(key)
if err != nil {
return 0, err
}
c.mu.Lock()
// Remove any prior entry first; we'll re-link after the file is
// written successfully.
if e, ok := c.items[key]; ok {
c.unlink(e)
c.curBytes -= e.size
_ = os.Remove(p)
delete(c.items, key)
}
c.mu.Unlock()
//nolint:gosec // G304: path() rejects keys with a separator
f, err := os.OpenFile(
p, os.O_CREATE|os.O_TRUNC|os.O_WRONLY, blobCacheFileMode)
if err != nil {
return 0, fmt.Errorf("creating cache file: %w", err)
}
written, copyErr := io.Copy(f, r)
closeErr := f.Close()
if copyErr != nil {
_ = os.Remove(p)
return written, fmt.Errorf("streaming to cache file: %w", copyErr)
}
if closeErr != nil {
_ = os.Remove(p)
return written, fmt.Errorf("closing cache file: %w", closeErr)
}
c.mu.Lock()
defer c.mu.Unlock()
// If the entry would exceed maxBytes outright, drop it on the
// floor — but the restore path passes math.MaxInt64 as maxBytes
// so this branch is effectively unreachable there.
if written > c.maxBytes {
_ = os.Remove(p)
return written, nil
}
e := &blobDiskCacheEntry{key: key, size: written}
c.pushFront(e)
c.items[key] = e
c.curBytes += written
for c.curBytes > c.maxBytes && c.tail != nil {
c.evictLRU()
}
if n := len(c.items); n > c.peakLen {
c.peakLen = n
}
return written, nil
}
// Get reads a cached blob from disk. Returns data and true on hit.
func (c *blobDiskCache) Get(key string) ([]byte, bool) {
p, err := c.path(key)
if err != nil {
return nil, false
}
c.mu.Lock()
c.getCalls++
e, ok := c.items[key]
if !ok {
c.mu.Unlock()
return nil, false
}
c.unlink(e)
c.pushFront(e)
c.mu.Unlock()
//nolint:gosec // G304: path() rejects keys with a separator
data, err := os.ReadFile(p)
if err != nil {
c.mu.Lock()
if e2, ok2 := c.items[key]; ok2 && e2 == e {
c.unlink(e)
delete(c.items, key)
c.curBytes -= e.size
}
c.mu.Unlock()
return nil, false
}
return data, true
}
// ReadAt reads a slice of a cached blob without loading the entire blob into memory.
func (c *blobDiskCache) ReadAt(key string, offset, length int64) ([]byte, error) {
p, err := c.path(key)
if err != nil {
return nil, err
}
// offset and length come from a blob_chunks row read back from the
// destination. A negative value must be rejected outright; the upper
// bound is checked as length > size-offset (a subtraction) so a huge
// offset+length cannot overflow int64 and slip past the check.
if offset < 0 || length < 0 {
return nil, fmt.Errorf("%w: offset=%d length=%d",
errCacheNegativeRead, offset, length)
}
c.mu.Lock()
c.readAtCalls++
e, ok := c.items[key]
if !ok {
c.mu.Unlock()
return nil, fmt.Errorf("%w: %q", errCacheKeyMissing, key)
}
if length > e.size-offset {
c.mu.Unlock()
return nil, fmt.Errorf("%w: offset=%d length=%d size=%d",
errCacheReadBeyondBlob, offset, length, e.size)
}
c.unlink(e)
c.pushFront(e)
c.mu.Unlock()
f, err := os.Open(p) //nolint:gosec // G304: path() rejects keys with a separator
if err != nil {
return nil, err
}
defer func() { _ = f.Close() }()
buf := make([]byte, length)
_, err = f.ReadAt(buf, offset)
if err != nil {
return nil, err
}
return buf, nil
}
// Has returns whether a key exists in the cache.
func (c *blobDiskCache) Has(key string) bool {
c.mu.Lock()
defer c.mu.Unlock()
_, ok := c.items[key]
return ok
}
// Delete removes a blob from the cache and its disk file. No-op if absent.
// Used by restore's sweep logic to free blobs whose chunks have all been
// restored (so they will never be needed again during this restore).
func (c *blobDiskCache) Delete(key string) {
c.mu.Lock()
defer c.mu.Unlock()
e, ok := c.items[key]
if !ok {
return
}
c.unlink(e)
delete(c.items, key)
c.curBytes -= e.size
// The key is already in the map, so it passed path() when it was
// inserted; the error cannot occur here.
p, err := c.path(key)
if err == nil {
_ = os.Remove(p)
}
}
// Keys returns a snapshot of all cached keys. Safe for iteration without
// holding the cache lock; the cache may change concurrently.
func (c *blobDiskCache) Keys() []string {
c.mu.Lock()
defer c.mu.Unlock()
keys := make([]string, 0, len(c.items))
for k := range c.items {
keys = append(keys, k)
}
return keys
}
// Size returns current total cached bytes.
func (c *blobDiskCache) Size() int64 {
c.mu.Lock()
defer c.mu.Unlock()
return c.curBytes
}
// Len returns number of cached entries.
func (c *blobDiskCache) Len() int {
c.mu.Lock()
defer c.mu.Unlock()
return len(c.items)
}
// GetCalls returns the number of times Get has been called.
func (c *blobDiskCache) GetCalls() int {
c.mu.Lock()
defer c.mu.Unlock()
return c.getCalls
}
// ReadAtCalls returns the number of times ReadAt has been called.
func (c *blobDiskCache) ReadAtCalls() int {
c.mu.Lock()
defer c.mu.Unlock()
return c.readAtCalls
}
// PeakLen returns the maximum number of cached entries ever held at
// once during this cache's lifetime.
func (c *blobDiskCache) PeakLen() int {
c.mu.Lock()
defer c.mu.Unlock()
return c.peakLen
}
// Close removes the cache directory and all cached blobs.
func (c *blobDiskCache) Close() error {
c.mu.Lock()
defer c.mu.Unlock()
c.items = nil
c.head = nil
c.tail = nil
c.curBytes = 0
return os.RemoveAll(c.dir)
}
// path returns the on-disk location of the cache file for key. The key is
// a blob hash read back from the destination and is not trusted: a value
// such as "aa/../../../home/u/.profile" would otherwise make filepath.Join
// escape the cache directory, so a key containing a path separator is
// refused rather than joined.
func (c *blobDiskCache) path(key string) (string, error) {
if strings.ContainsRune(key, '/') ||
strings.ContainsRune(key, filepath.Separator) {
return "", fmt.Errorf("%w: %q", errCacheKeyHasSeparator, key)
}
return filepath.Join(c.dir, key), nil
}
func (c *blobDiskCache) unlink(e *blobDiskCacheEntry) {
if e.prev != nil {
e.prev.next = e.next
} else {
c.head = e.next
}
if e.next != nil {
e.next.prev = e.prev
} else {
c.tail = e.prev
}
e.prev = nil
e.next = nil
}
func (c *blobDiskCache) pushFront(e *blobDiskCacheEntry) {
e.prev = nil
e.next = c.head
if c.head != nil {
c.head.prev = e
}
c.head = e
if c.tail == nil {
c.tail = e
}
}
func (c *blobDiskCache) evictLRU() {
if c.tail == nil {
return
}
victim := c.tail
c.unlink(victim)
delete(c.items, victim.key)
c.curBytes -= victim.size
// victim.key was validated by path() on insertion, so this cannot err.
p, err := c.path(victim.key)
if err == nil {
_ = os.Remove(p)
}
}