Count each file, byte and upload once in backup statistics (closes #225)
check / check (push) Waiting to run

The scanner added a changed file's bytes again for each new chunk and
counted a file as unchanged for each chunk already stored. It now
counts files and bytes once, in the scan phase, and counts its own
uploads, so a --cron run, which has no progress reporter, records
them. The blob count no longer adds earlier paths' blobs again.

The snapshots row now stores the size of all files in total_size and
the referenced blobs' sizes in blob_size, blob_uncompressed_size and
compression_ratio, as docs/DATAMODEL.md says. Those sizes come from
one query, and a failed query fails the snapshot. DATAMODEL.md now
says chunk_count and blob_count count what the run added.

Removed UpdateSnapshotStats and GetCountBySnapshot, which nothing
calls any more.

Model: opus-5-5
This commit is contained in:
2026-10-07 00:00:51 +00:00
parent 85d4ef118d
commit 80b44d1c16
12 changed files with 471 additions and 172 deletions
+43 -67
View File
@@ -92,9 +92,6 @@ type Scanner struct {
// Mutex for coordinating blob creation
packerMu sync.Mutex // Blocks chunk production during blob creation
// Context for cancellation
scanCtx context.Context //nolint:containedctx // set per-Scan for packer callbacks
}
// Periodic status output intervals and thresholds for the scan and
@@ -134,18 +131,23 @@ type ScannerConfig struct {
SkipErrors bool
}
// ScanResult contains the results of a scan operation
// ScanResult contains the results of a scan operation. Files and bytes
// are counted per file: BytesScanned is the size of the new and changed
// files, BytesSkipped that of the unchanged ones.
type ScanResult struct {
FilesScanned int
FilesSkipped int
FilesDeleted int
BytesScanned int64
BytesSkipped int64
BytesDeleted int64
ChunksCreated int
BlobsCreated int
StartTime time.Time
EndTime time.Time
FilesScanned int
FilesSkipped int
FilesDeleted int
BytesScanned int64
BytesSkipped int64
BytesDeleted int64
ChunksCreated int
BlobsCreated int
BlobsUploaded int
BytesUploaded int64
UploadDuration time.Duration
StartTime time.Time
EndTime time.Time
}
// NewScanner creates a new scanner instance
@@ -211,7 +213,6 @@ func (s *Scanner) Scan(
s.snapshotID = snapshotID
// Store source path for file records (used during restore)
s.currentSourcePath = path
s.scanCtx = ctx
result := &ScanResult{
StartTime: time.Now().UTC(),
}
@@ -219,7 +220,9 @@ func (s *Scanner) Scan(
// Set blob handler for concurrent upload
if s.storage != nil {
log.Debug("Setting blob handler for storage uploads")
s.packer.SetBlobHandler(s.handleBlobReady)
s.packer.SetBlobHandler(func(blobWithReader *blob.WithReader) error {
return s.handleBlobReady(ctx, blobWithReader, result)
})
} else {
log.Debug("No storage configured, blobs will not be uploaded")
}
@@ -288,8 +291,7 @@ func (s *Scanner) Scan(
log.Info("Phase 2/3: Skipping (no files need processing, metadata-only snapshot)")
}
// Finalize result with blob statistics
s.finalizeScanResult(ctx, result)
result.EndTime = time.Now().UTC()
return result, nil
}
@@ -431,27 +433,6 @@ func (s *Scanner) summarizeScanPhase(
s.ui.Completef("%s.", msg)
}
// finalizeScanResult populates final blob statistics in the scan result
// by querying the packer and database for blob/upload counts
func (s *Scanner) finalizeScanResult(ctx context.Context, result *ScanResult) {
blobs := s.packer.GetFinishedBlobs()
result.BlobsCreated += len(blobs)
// Query database for actual blob count created during this snapshot
// The database is authoritative, especially for concurrent blob uploads
// We count uploads rather than all snapshot_blobs to get only NEW blobs
if s.snapshotID != "" {
uploadCount, err := s.repos.Uploads.GetCountBySnapshot(ctx, s.snapshotID)
if err != nil {
log.Warn("Failed to query upload count from database", "error", err)
} else {
result.BlobsCreated = int(uploadCount)
}
}
result.EndTime = time.Now().UTC()
}
// loadKnownFiles loads the known files at and beneath path from the
// database into a map for fast lookup. Every loaded file the scan does
// not find is counted as deleted. This avoids per-file database queries
@@ -1511,24 +1492,24 @@ func (s *Scanner) finalizeProcessPhase(ctx context.Context, result *ScanResult)
}
// handleBlobReady is called by the packer when a blob is finalized
func (s *Scanner) handleBlobReady(blobWithReader *blob.WithReader) error {
func (s *Scanner) handleBlobReady(
ctx context.Context, blobWithReader *blob.WithReader, result *ScanResult,
) error {
startTime := time.Now().UTC()
finishedBlob := blobWithReader.FinishedBlob
result.BlobsCreated++
if s.progress != nil {
s.progress.ReportUploadStart(finishedBlob.Hash, finishedBlob.Compressed)
s.progress.GetStats().BlobsCreated.Add(1)
}
ctx := s.scanCtx
if ctx == nil {
ctx = context.Background()
}
blobPath := fmt.Sprintf("blobs/%s/%s/%s",
finishedBlob.Hash[:2], finishedBlob.Hash[2:4], finishedBlob.Hash)
blobExists, err := s.uploadBlobIfNeeded(ctx, blobPath, blobWithReader, startTime)
blobExists, err := s.uploadBlobIfNeeded(
ctx, blobPath, blobWithReader, startTime, result)
if err != nil {
s.cleanupBlobTempFile(blobWithReader)
@@ -1563,6 +1544,7 @@ func (s *Scanner) uploadBlobIfNeeded(
blobPath string,
blobWithReader *blob.WithReader,
startTime time.Time,
result *ScanResult,
) (bool, error) {
finishedBlob := blobWithReader.FinishedBlob
@@ -1598,6 +1580,10 @@ func (s *Scanner) uploadBlobIfNeeded(
uploadDuration := time.Since(startTime)
uploadSpeedBps := float64(finishedBlob.Compressed) / uploadDuration.Seconds()
result.BlobsUploaded++
result.BytesUploaded += finishedBlob.Compressed
result.UploadDuration += uploadDuration
s.ui.Completef("Uploaded blob %s (%s) in %s at %s.",
s.ui.Hex(finishedBlob.Hash),
s.ui.Size(finishedBlob.Compressed),
@@ -1813,9 +1799,9 @@ func (s *Scanner) processFileStreaming(
size: chunk.Size,
})
s.updateChunkStats(chunkExists, chunk.Size, result)
if !chunkExists {
s.updateChunkStats(chunk.Size, result)
err := s.addChunkToPacker(ctx, chunk)
if err != nil {
// Mark as a packer error so --skip-errors cannot swallow it:
@@ -1843,26 +1829,16 @@ func (s *Scanner) processFileStreaming(
return nil
}
// updateChunkStats updates scan result and progress stats for a processed chunk
func (s *Scanner) updateChunkStats(
chunkExists bool, chunkSize int64, result *ScanResult,
) {
if chunkExists {
result.FilesSkipped++
// updateChunkStats counts a chunk that was not already stored. The scan
// result's file counts, BytesScanned and BytesSkipped are not touched
// here: the scan phase counts each file once.
func (s *Scanner) updateChunkStats(chunkSize int64, result *ScanResult) {
result.ChunksCreated++
result.BytesSkipped += chunkSize
if s.progress != nil {
s.progress.GetStats().BytesSkipped.Add(chunkSize)
}
} else {
result.ChunksCreated++
result.BytesScanned += chunkSize
if s.progress != nil {
s.progress.GetStats().ChunksCreated.Add(1)
s.progress.GetStats().BytesProcessed.Add(chunkSize)
s.progress.UpdateChunkingActivity()
}
if s.progress != nil {
s.progress.GetStats().ChunksCreated.Add(1)
s.progress.GetStats().BytesProcessed.Add(chunkSize)
s.progress.UpdateChunkingActivity()
}
}