check / check (push) Successful in 14m1s
The scanner added a changed file's bytes again for each new chunk and counted a file as unchanged for each chunk already stored. It now counts files and bytes once, in the scan phase, and counts its own uploads, so a --cron run, which has no progress reporter, records them. The blob count no longer adds earlier paths' blobs again. The snapshots row now stores the size of all files in total_size and the referenced blobs' sizes in blob_size, blob_uncompressed_size and compression_ratio, as docs/DATAMODEL.md says; snapshot_blobs is filled before the stats for that. DATAMODEL.md now says chunk_count and blob_count count what the run added. Removed the uncalled UpdateSnapshotStats, which stored bytes uploaded in blob_size. Model: opus-5-5
133 lines
4.8 KiB
Go
133 lines
4.8 KiB
Go
package database
|
|
|
|
import (
|
|
"time"
|
|
|
|
"sneak.berlin/go/vaultik/internal/types"
|
|
)
|
|
|
|
// File represents a file or directory in the backup system.
|
|
// It stores metadata about files including timestamps, permissions, ownership,
|
|
// and symlink targets. This information is used to restore files with their
|
|
// original attributes.
|
|
type File struct {
|
|
ID types.FileID // UUID primary key
|
|
Path types.FilePath // Absolute path of the file
|
|
|
|
// SourcePath is the source directory this file came from (used for
|
|
// restore path stripping).
|
|
SourcePath types.SourcePath
|
|
MTime time.Time
|
|
Size int64
|
|
Mode uint32
|
|
UID uint32
|
|
GID uint32
|
|
LinkTarget types.FilePath // empty for regular files, target path for symlinks
|
|
}
|
|
|
|
// IsSymlink returns true if this file is a symbolic link.
|
|
// A file is considered a symlink if it has a non-empty LinkTarget.
|
|
func (f *File) IsSymlink() bool {
|
|
return f.LinkTarget != ""
|
|
}
|
|
|
|
// FileChunk represents the mapping between files and their constituent chunks.
|
|
// Large files are split into multiple chunks for efficient deduplication and storage.
|
|
// The Idx field maintains the order of chunks within a file.
|
|
type FileChunk struct {
|
|
FileID types.FileID
|
|
Idx int
|
|
ChunkHash types.ChunkHash
|
|
}
|
|
|
|
// Chunk represents a data chunk in the deduplication system.
|
|
// Files are split into chunks which are content-addressed by their hash.
|
|
// The ChunkHash is the SHA256 hash of the chunk content, used for deduplication.
|
|
type Chunk struct {
|
|
ChunkHash types.ChunkHash
|
|
Size int64
|
|
}
|
|
|
|
// Blob represents a blob record in the database.
|
|
// A blob is Vaultik's final storage unit - a large file (up to 10GB) containing
|
|
// many compressed and encrypted chunks from multiple source files.
|
|
// Blobs are content-addressed: the filename in S3 is
|
|
// hex(SHA256(SHA256(uncompressed blob contents))), computed from the chunk data
|
|
// before compression and encryption (not from the stored bytes). See
|
|
// blobgen.DoubleSHA256 and docs/REPOSTRUCTURE.md.
|
|
type Blob struct {
|
|
ID types.BlobID // UUID assigned when blob creation starts
|
|
|
|
// Hash is hex(SHA256(SHA256(uncompressed blob contents)))
|
|
// (empty until finalized); see the type comment above.
|
|
Hash types.BlobHash
|
|
CreatedTS time.Time // When blob creation started
|
|
FinishedTS *time.Time // When blob was finalized (nil if still packing)
|
|
UncompressedSize int64 // Total size of raw chunks before compression
|
|
CompressedSize int64 // Size after compression and encryption
|
|
UploadedTS *time.Time // When blob was uploaded to S3 (nil if not uploaded)
|
|
}
|
|
|
|
// BlobChunk represents the mapping between blobs and the chunks they contain.
|
|
// This allows tracking which chunks are stored in which blobs, along with
|
|
// their position and size within the blob. The offset and length fields
|
|
// enable extracting specific chunks from a blob without processing the entire blob.
|
|
type BlobChunk struct {
|
|
BlobID types.BlobID
|
|
ChunkHash types.ChunkHash
|
|
Offset int64
|
|
Length int64
|
|
}
|
|
|
|
// ChunkFile represents the reverse mapping showing which files contain a
|
|
// specific chunk. This is used during deduplication to identify all files
|
|
// that share a chunk, which is important for garbage collection and
|
|
// integrity verification.
|
|
type ChunkFile struct {
|
|
ChunkHash types.ChunkHash
|
|
FileID types.FileID
|
|
FileOffset int64
|
|
Length int64
|
|
}
|
|
|
|
// Snapshot represents a snapshot record in the database
|
|
type Snapshot struct {
|
|
ID types.SnapshotID
|
|
Hostname types.Hostname
|
|
VaultikVersion types.Version
|
|
VaultikGitRevision types.GitRevision
|
|
StartedAt time.Time
|
|
CompletedAt *time.Time // nil if still in progress
|
|
FileCount int64
|
|
ChunkCount int64 // Chunks this snapshot stored that were not stored before
|
|
BlobCount int64 // Blobs this snapshot created
|
|
TotalSize int64 // Total size of all referenced files
|
|
|
|
// BlobSize is the total size of all referenced blobs (compressed and
|
|
// encrypted).
|
|
BlobSize int64
|
|
BlobUncompressedSize int64 // Total uncompressed size of all referenced blobs
|
|
CompressionRatio float64 // Compression ratio (BlobSize / BlobUncompressedSize)
|
|
CompressionLevel int // Compression level used for this snapshot
|
|
UploadBytes int64 // Total bytes uploaded during this snapshot
|
|
UploadDurationMs int64 // Total milliseconds spent uploading to S3
|
|
}
|
|
|
|
// IsComplete returns true if the snapshot has completed
|
|
func (s *Snapshot) IsComplete() bool {
|
|
return s.CompletedAt != nil
|
|
}
|
|
|
|
// SnapshotFile represents the mapping between snapshots and files
|
|
type SnapshotFile struct {
|
|
SnapshotID types.SnapshotID
|
|
FileID types.FileID
|
|
}
|
|
|
|
// SnapshotBlob represents the mapping between snapshots and blobs
|
|
type SnapshotBlob struct {
|
|
SnapshotID types.SnapshotID
|
|
BlobID types.BlobID
|
|
BlobHash types.BlobHash // Denormalized for easier manifest generation
|
|
}
|