Count each file, byte and upload once in backup statistics (closes #225)
check / check (push) Canceled after 0s
check / check (push) Canceled after 0s
The scanner added a changed file's bytes again for each new chunk and counted a file as unchanged for each chunk already stored. It now counts files and bytes once, in the scan phase, and counts its own uploads, so a --cron run, which has no progress reporter, records them. The blob count no longer adds earlier paths' blobs again. The snapshots row now stores the size of all files in total_size and the referenced blobs' sizes in blob_size, blob_uncompressed_size and compression_ratio, as docs/DATAMODEL.md says. Those sizes come from one query, and a failed query fails the snapshot. DATAMODEL.md now says chunk_count and blob_count count what the run added. Removed UpdateSnapshotStats and GetCountBySnapshot, which nothing calls any more. Model: opus-5-5
This commit was merged in pull request #254.
This commit is contained in:
@@ -66,7 +66,6 @@ type ProgressStats struct {
|
||||
BlobsCreated atomic.Int64
|
||||
BlobsUploaded atomic.Int64
|
||||
BytesUploaded atomic.Int64
|
||||
UploadDurationMs atomic.Int64 // Total milliseconds spent uploading
|
||||
CurrentFile atomic.Value // stores string
|
||||
TotalSize atomic.Int64 // Total size to process (set after scan phase)
|
||||
TotalFiles atomic.Int64 // Total files to process in phase 2
|
||||
@@ -231,9 +230,6 @@ func (pr *ProgressReporter) ReportUploadComplete(
|
||||
// Clear current upload
|
||||
pr.stats.CurrentUpload.Store((*UploadInfo)(nil))
|
||||
|
||||
// Add to total upload duration
|
||||
pr.stats.UploadDurationMs.Add(duration.Milliseconds())
|
||||
|
||||
// Calculate speed
|
||||
if duration < time.Millisecond {
|
||||
duration = time.Millisecond
|
||||
|
||||
@@ -92,9 +92,6 @@ type Scanner struct {
|
||||
|
||||
// Mutex for coordinating blob creation
|
||||
packerMu sync.Mutex // Blocks chunk production during blob creation
|
||||
|
||||
// Context for cancellation
|
||||
scanCtx context.Context //nolint:containedctx // set per-Scan for packer callbacks
|
||||
}
|
||||
|
||||
// Periodic status output intervals and thresholds for the scan and
|
||||
@@ -134,18 +131,23 @@ type ScannerConfig struct {
|
||||
SkipErrors bool
|
||||
}
|
||||
|
||||
// ScanResult contains the results of a scan operation
|
||||
// ScanResult contains the results of a scan operation. Files and bytes
|
||||
// are counted per file: BytesScanned is the size of the new and changed
|
||||
// files, BytesSkipped that of the unchanged ones.
|
||||
type ScanResult struct {
|
||||
FilesScanned int
|
||||
FilesSkipped int
|
||||
FilesDeleted int
|
||||
BytesScanned int64
|
||||
BytesSkipped int64
|
||||
BytesDeleted int64
|
||||
ChunksCreated int
|
||||
BlobsCreated int
|
||||
StartTime time.Time
|
||||
EndTime time.Time
|
||||
FilesScanned int
|
||||
FilesSkipped int
|
||||
FilesDeleted int
|
||||
BytesScanned int64
|
||||
BytesSkipped int64
|
||||
BytesDeleted int64
|
||||
ChunksCreated int
|
||||
BlobsCreated int
|
||||
BlobsUploaded int
|
||||
BytesUploaded int64
|
||||
UploadDuration time.Duration
|
||||
StartTime time.Time
|
||||
EndTime time.Time
|
||||
}
|
||||
|
||||
// NewScanner creates a new scanner instance
|
||||
@@ -211,7 +213,6 @@ func (s *Scanner) Scan(
|
||||
s.snapshotID = snapshotID
|
||||
// Store source path for file records (used during restore)
|
||||
s.currentSourcePath = path
|
||||
s.scanCtx = ctx
|
||||
result := &ScanResult{
|
||||
StartTime: time.Now().UTC(),
|
||||
}
|
||||
@@ -219,7 +220,9 @@ func (s *Scanner) Scan(
|
||||
// Set blob handler for concurrent upload
|
||||
if s.storage != nil {
|
||||
log.Debug("Setting blob handler for storage uploads")
|
||||
s.packer.SetBlobHandler(s.handleBlobReady)
|
||||
s.packer.SetBlobHandler(func(blobWithReader *blob.WithReader) error {
|
||||
return s.handleBlobReady(ctx, blobWithReader, result)
|
||||
})
|
||||
} else {
|
||||
log.Debug("No storage configured, blobs will not be uploaded")
|
||||
}
|
||||
@@ -288,8 +291,7 @@ func (s *Scanner) Scan(
|
||||
log.Info("Phase 2/3: Skipping (no files need processing, metadata-only snapshot)")
|
||||
}
|
||||
|
||||
// Finalize result with blob statistics
|
||||
s.finalizeScanResult(ctx, result)
|
||||
result.EndTime = time.Now().UTC()
|
||||
|
||||
return result, nil
|
||||
}
|
||||
@@ -431,27 +433,6 @@ func (s *Scanner) summarizeScanPhase(
|
||||
s.ui.Completef("%s.", msg)
|
||||
}
|
||||
|
||||
// finalizeScanResult populates final blob statistics in the scan result
|
||||
// by querying the packer and database for blob/upload counts
|
||||
func (s *Scanner) finalizeScanResult(ctx context.Context, result *ScanResult) {
|
||||
blobs := s.packer.GetFinishedBlobs()
|
||||
result.BlobsCreated += len(blobs)
|
||||
|
||||
// Query database for actual blob count created during this snapshot
|
||||
// The database is authoritative, especially for concurrent blob uploads
|
||||
// We count uploads rather than all snapshot_blobs to get only NEW blobs
|
||||
if s.snapshotID != "" {
|
||||
uploadCount, err := s.repos.Uploads.GetCountBySnapshot(ctx, s.snapshotID)
|
||||
if err != nil {
|
||||
log.Warn("Failed to query upload count from database", "error", err)
|
||||
} else {
|
||||
result.BlobsCreated = int(uploadCount)
|
||||
}
|
||||
}
|
||||
|
||||
result.EndTime = time.Now().UTC()
|
||||
}
|
||||
|
||||
// loadKnownFiles loads the known files at and beneath path from the
|
||||
// database into a map for fast lookup. Every loaded file the scan does
|
||||
// not find is counted as deleted. This avoids per-file database queries
|
||||
@@ -1511,24 +1492,24 @@ func (s *Scanner) finalizeProcessPhase(ctx context.Context, result *ScanResult)
|
||||
}
|
||||
|
||||
// handleBlobReady is called by the packer when a blob is finalized
|
||||
func (s *Scanner) handleBlobReady(blobWithReader *blob.WithReader) error {
|
||||
func (s *Scanner) handleBlobReady(
|
||||
ctx context.Context, blobWithReader *blob.WithReader, result *ScanResult,
|
||||
) error {
|
||||
startTime := time.Now().UTC()
|
||||
finishedBlob := blobWithReader.FinishedBlob
|
||||
|
||||
result.BlobsCreated++
|
||||
|
||||
if s.progress != nil {
|
||||
s.progress.ReportUploadStart(finishedBlob.Hash, finishedBlob.Compressed)
|
||||
s.progress.GetStats().BlobsCreated.Add(1)
|
||||
}
|
||||
|
||||
ctx := s.scanCtx
|
||||
if ctx == nil {
|
||||
ctx = context.Background()
|
||||
}
|
||||
|
||||
blobPath := fmt.Sprintf("blobs/%s/%s/%s",
|
||||
finishedBlob.Hash[:2], finishedBlob.Hash[2:4], finishedBlob.Hash)
|
||||
|
||||
blobExists, err := s.uploadBlobIfNeeded(ctx, blobPath, blobWithReader, startTime)
|
||||
blobExists, err := s.uploadBlobIfNeeded(
|
||||
ctx, blobPath, blobWithReader, startTime, result)
|
||||
if err != nil {
|
||||
s.cleanupBlobTempFile(blobWithReader)
|
||||
|
||||
@@ -1563,6 +1544,7 @@ func (s *Scanner) uploadBlobIfNeeded(
|
||||
blobPath string,
|
||||
blobWithReader *blob.WithReader,
|
||||
startTime time.Time,
|
||||
result *ScanResult,
|
||||
) (bool, error) {
|
||||
finishedBlob := blobWithReader.FinishedBlob
|
||||
|
||||
@@ -1598,6 +1580,10 @@ func (s *Scanner) uploadBlobIfNeeded(
|
||||
uploadDuration := time.Since(startTime)
|
||||
uploadSpeedBps := float64(finishedBlob.Compressed) / uploadDuration.Seconds()
|
||||
|
||||
result.BlobsUploaded++
|
||||
result.BytesUploaded += finishedBlob.Compressed
|
||||
result.UploadDuration += uploadDuration
|
||||
|
||||
s.ui.Completef("Uploaded blob %s (%s) in %s at %s.",
|
||||
s.ui.Hex(finishedBlob.Hash),
|
||||
s.ui.Size(finishedBlob.Compressed),
|
||||
@@ -1813,9 +1799,9 @@ func (s *Scanner) processFileStreaming(
|
||||
size: chunk.Size,
|
||||
})
|
||||
|
||||
s.updateChunkStats(chunkExists, chunk.Size, result)
|
||||
|
||||
if !chunkExists {
|
||||
s.updateChunkStats(chunk.Size, result)
|
||||
|
||||
err := s.addChunkToPacker(ctx, chunk)
|
||||
if err != nil {
|
||||
// Mark as a packer error so --skip-errors cannot swallow it:
|
||||
@@ -1843,26 +1829,16 @@ func (s *Scanner) processFileStreaming(
|
||||
return nil
|
||||
}
|
||||
|
||||
// updateChunkStats updates scan result and progress stats for a processed chunk
|
||||
func (s *Scanner) updateChunkStats(
|
||||
chunkExists bool, chunkSize int64, result *ScanResult,
|
||||
) {
|
||||
if chunkExists {
|
||||
result.FilesSkipped++
|
||||
// updateChunkStats counts a chunk that was not already stored. The scan
|
||||
// result's file counts, BytesScanned and BytesSkipped are not touched
|
||||
// here: the scan phase counts each file once.
|
||||
func (s *Scanner) updateChunkStats(chunkSize int64, result *ScanResult) {
|
||||
result.ChunksCreated++
|
||||
|
||||
result.BytesSkipped += chunkSize
|
||||
if s.progress != nil {
|
||||
s.progress.GetStats().BytesSkipped.Add(chunkSize)
|
||||
}
|
||||
} else {
|
||||
result.ChunksCreated++
|
||||
result.BytesScanned += chunkSize
|
||||
|
||||
if s.progress != nil {
|
||||
s.progress.GetStats().ChunksCreated.Add(1)
|
||||
s.progress.GetStats().BytesProcessed.Add(chunkSize)
|
||||
s.progress.UpdateChunkingActivity()
|
||||
}
|
||||
if s.progress != nil {
|
||||
s.progress.GetStats().ChunksCreated.Add(1)
|
||||
s.progress.GetStats().BytesProcessed.Add(chunkSize)
|
||||
s.progress.UpdateChunkingActivity()
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -154,26 +154,6 @@ func (sm *SnapshotManager) CreateSnapshotWithName(
|
||||
return snapshotID, nil
|
||||
}
|
||||
|
||||
// UpdateSnapshotStats updates the statistics for a snapshot during backup
|
||||
func (sm *SnapshotManager) UpdateSnapshotStats(
|
||||
ctx context.Context, snapshotID string, stats BackupStats,
|
||||
) error {
|
||||
err := sm.repos.WithTx(ctx, func(ctx context.Context, tx *sql.Tx) error {
|
||||
return sm.repos.Snapshots.UpdateCounts(ctx, tx, snapshotID,
|
||||
int64(stats.FilesScanned),
|
||||
int64(stats.ChunksCreated),
|
||||
int64(stats.BlobsCreated),
|
||||
stats.BytesScanned,
|
||||
stats.BytesUploaded,
|
||||
)
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("updating snapshot stats: %w", err)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// UpdateSnapshotStatsExtended updates snapshot statistics with extended metrics.
|
||||
// This includes compression level, uncompressed blob size, and upload duration.
|
||||
func (sm *SnapshotManager) UpdateSnapshotStatsExtended(
|
||||
@@ -185,8 +165,8 @@ func (sm *SnapshotManager) UpdateSnapshotStatsExtended(
|
||||
int64(stats.FilesScanned),
|
||||
int64(stats.ChunksCreated),
|
||||
int64(stats.BlobsCreated),
|
||||
stats.BytesScanned,
|
||||
stats.BytesUploaded,
|
||||
stats.TotalSize,
|
||||
stats.BlobSize,
|
||||
)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -196,6 +176,7 @@ func (sm *SnapshotManager) UpdateSnapshotStatsExtended(
|
||||
return sm.repos.Snapshots.UpdateExtendedStats(ctx, tx, snapshotID,
|
||||
stats.BlobUncompressedSize,
|
||||
stats.CompressionLevel,
|
||||
stats.BytesUploaded,
|
||||
stats.UploadDurationMs,
|
||||
)
|
||||
})
|
||||
@@ -890,7 +871,7 @@ func (sm *SnapshotManager) getFileSize(path string) int64 {
|
||||
// BackupStats contains statistics from a backup operation
|
||||
type BackupStats struct {
|
||||
FilesScanned int
|
||||
BytesScanned int64
|
||||
TotalSize int64 // Total size of all files examined
|
||||
ChunksCreated int
|
||||
BlobsCreated int
|
||||
BytesUploaded int64
|
||||
@@ -900,6 +881,7 @@ type BackupStats struct {
|
||||
type ExtendedBackupStats struct {
|
||||
BackupStats
|
||||
|
||||
BlobSize int64 // Total compressed size of all referenced blobs
|
||||
BlobUncompressedSize int64 // Total uncompressed size of all referenced blobs
|
||||
CompressionLevel int // Compression level used for this snapshot
|
||||
UploadDurationMs int64 // Total milliseconds spent uploading to S3
|
||||
|
||||
Reference in New Issue
Block a user