Re-chunk a known file whose chunks no uploaded blob holds (closes #214)
File rows are shared by every snapshot and updated in place, while a blob row is deleted once no snapshot references it. Removing the newest snapshot, or the prune after an interrupted run, could drop the only blob holding a changed file's current chunks while an older snapshot kept the file row. The next backup compared metadata only, skipped the file, and completed a snapshot that could not restore it. The scanner now loads the IDs of known files that list a chunk no uploaded blob holds and re-chunks them even when their metadata is unchanged. The tests append to a file, so the file keeps its first chunk in a blob the first snapshot still references. Each backup run gets its own snapshot name, so the second-precision snapshot IDs differ without sleeping. Model: opus-5-5
This commit was merged in pull request #234.
This commit is contained in:
@@ -73,6 +73,11 @@ type Scanner struct {
|
||||
knownChunks map[string]struct{}
|
||||
knownChunksMu sync.RWMutex
|
||||
|
||||
// filesToRechunk holds the IDs of known files that list a chunk no
|
||||
// uploaded blob holds; they are re-chunked even when their metadata
|
||||
// is unchanged.
|
||||
filesToRechunk map[types.FileID]struct{}
|
||||
|
||||
// Pending chunk hashes - chunks that have been added to packer but not
|
||||
// yet committed to DB. When a blob finalizes, the committed chunks are
|
||||
// removed from this set.
|
||||
@@ -325,9 +330,42 @@ func (s *Scanner) loadDatabaseState(
|
||||
s.ui.Completef("Loaded %s known chunks from local index database.",
|
||||
s.ui.Count(len(s.knownChunks)))
|
||||
|
||||
err = s.loadFilesToRechunk(ctx, path)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("loading files to re-chunk: %w", err)
|
||||
}
|
||||
|
||||
return knownFiles, nil
|
||||
}
|
||||
|
||||
// loadFilesToRechunk loads the IDs of known files under path that list a
|
||||
// chunk no uploaded blob holds. A file row is shared by every snapshot
|
||||
// that lists the file and is updated in place when the file changes,
|
||||
// while a blob row is deleted once no snapshot references it. Removing
|
||||
// the only snapshot that references a changed file's current blob
|
||||
// therefore leaves an older snapshot keeping a file row whose metadata
|
||||
// matches the disk while no blob the local index records as uploaded
|
||||
// holds its chunks, so a new snapshot cannot reference them. The dropped
|
||||
// blob can still be in remote storage until prune removes it.
|
||||
func (s *Scanner) loadFilesToRechunk(ctx context.Context, path string) error {
|
||||
ids, err := s.repos.Files.ListIDsWithChunksNotInUploadedBlobs(ctx, path)
|
||||
if err != nil {
|
||||
return fmt.Errorf("listing files: %w", err)
|
||||
}
|
||||
|
||||
s.filesToRechunk = make(map[types.FileID]struct{}, len(ids))
|
||||
for _, id := range ids {
|
||||
s.filesToRechunk[id] = struct{}{}
|
||||
}
|
||||
|
||||
if len(ids) > 0 {
|
||||
log.Info("Re-chunking known files whose chunks are not all "+
|
||||
"in uploaded blobs", "files", len(ids))
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// repairInterruptedBlobs discards blob rows left by a previous run whose
|
||||
// upload never completed. Such a blob has its chunks, blob_chunks, and
|
||||
// blobs rows committed to the local index before the upload is attempted,
|
||||
@@ -1208,6 +1246,11 @@ func (s *Scanner) checkFileInMemory(
|
||||
return file, true
|
||||
}
|
||||
|
||||
// No uploaded blob holds one of its chunks (see loadFilesToRechunk)
|
||||
if _, rechunk := s.filesToRechunk[fileID]; rechunk {
|
||||
return file, true
|
||||
}
|
||||
|
||||
// Check if file has changed
|
||||
if existingFile.Size != file.Size ||
|
||||
existingFile.MTime.Unix() != file.MTime.Unix() ||
|
||||
|
||||
Reference in New Issue
Block a user