Re-chunk a known file whose chunks no uploaded blob holds (closes #214)
check / check (pull_request) Successful in 8m16s
check / check (pull_request) Successful in 8m16s
File rows are shared by every snapshot and updated in place, while a blob row is deleted once no snapshot references it. Removing the newest snapshot, or the prune after an interrupted run, could drop the only blob holding a changed file's current chunks while an older snapshot kept the file row. The next backup compared metadata only, skipped the file, and completed a snapshot that could not restore it. The scanner now loads the IDs of known files that list a chunk no uploaded blob holds and re-chunks them even when their metadata is unchanged. The tests give each backup run its own snapshot name, so the second-precision snapshot IDs differ without sleeping. Model: opus-5-5
This commit is contained in:
@@ -266,6 +266,56 @@ func (r *FileRepository) ListByPrefix(
|
||||
return files, rows.Err()
|
||||
}
|
||||
|
||||
// ListIDsWithChunksNotInUploadedBlobs returns the IDs of the files whose
|
||||
// path starts with prefix and that list at least one chunk held by no
|
||||
// blob whose upload has completed (uploaded_ts set). A new snapshot
|
||||
// cannot reference such a chunk, so a backup must not treat the file as
|
||||
// unchanged even when its metadata matches the file on disk.
|
||||
func (r *FileRepository) ListIDsWithChunksNotInUploadedBlobs(
|
||||
ctx context.Context, prefix string,
|
||||
) ([]types.FileID, error) {
|
||||
query := `
|
||||
SELECT DISTINCT f.id
|
||||
FROM files f
|
||||
JOIN file_chunks fc ON fc.file_id = f.id
|
||||
WHERE f.path LIKE ? || '%'
|
||||
AND NOT EXISTS (
|
||||
SELECT 1
|
||||
FROM blob_chunks bc
|
||||
JOIN blobs b ON bc.blob_id = b.id
|
||||
WHERE bc.chunk_hash = fc.chunk_hash
|
||||
AND b.uploaded_ts IS NOT NULL
|
||||
)
|
||||
`
|
||||
|
||||
rows, err := r.db.conn.QueryContext(ctx, query, prefix)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("querying files: %w", err)
|
||||
}
|
||||
|
||||
defer func() {
|
||||
err := rows.Close()
|
||||
if err != nil {
|
||||
Fatalf("failed to close rows: %v", err)
|
||||
}
|
||||
}()
|
||||
|
||||
var ids []types.FileID
|
||||
|
||||
for rows.Next() {
|
||||
var id types.FileID
|
||||
|
||||
err := rows.Scan(&id)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("scanning file ID: %w", err)
|
||||
}
|
||||
|
||||
ids = append(ids, id)
|
||||
}
|
||||
|
||||
return ids, rows.Err()
|
||||
}
|
||||
|
||||
// ListAll returns all files in the database
|
||||
func (r *FileRepository) ListAll(ctx context.Context) ([]*File, error) {
|
||||
query := `
|
||||
|
||||
Reference in New Issue
Block a user