check / check (pull_request) Successful in 2m32s
Docs and comments only; no behaviour change. Corrects ten overclaims the security review found: snapshot names are hashed but the hash uses no secret, so a guessed hostname and name can be confirmed; a blob is named by hex(SHA256(SHA256(uncompressed contents))), stated once in docs/REPOSTRUCTURE.md and referenced elsewhere; double hashing does not hide known content (blob packing does); age uses ChaCha20-Poly1305, not XChaCha20; encryption is required, not optional; a snapshot is marked complete before its metadata is uploaded; the export comment now matches its only caller; deep verify detects corruption, not authorship; adding a recipient does not reach existing data; restore examples target a user-owned directory. Adds an Accepted Risks subsection under Security Considerations with the seven documented risks, cross-referenced from the README. Model: opus-4-8
133 lines
4.7 KiB
Go
133 lines
4.7 KiB
Go
package database
|
|
|
|
import (
|
|
"time"
|
|
|
|
"sneak.berlin/go/vaultik/internal/types"
|
|
)
|
|
|
|
// File represents a file or directory in the backup system.
|
|
// It stores metadata about files including timestamps, permissions, ownership,
|
|
// and symlink targets. This information is used to restore files with their
|
|
// original attributes.
|
|
type File struct {
|
|
ID types.FileID // UUID primary key
|
|
Path types.FilePath // Absolute path of the file
|
|
|
|
// SourcePath is the source directory this file came from (used for
|
|
// restore path stripping).
|
|
SourcePath types.SourcePath
|
|
MTime time.Time
|
|
Size int64
|
|
Mode uint32
|
|
UID uint32
|
|
GID uint32
|
|
LinkTarget types.FilePath // empty for regular files, target path for symlinks
|
|
}
|
|
|
|
// IsSymlink returns true if this file is a symbolic link.
|
|
// A file is considered a symlink if it has a non-empty LinkTarget.
|
|
func (f *File) IsSymlink() bool {
|
|
return f.LinkTarget != ""
|
|
}
|
|
|
|
// FileChunk represents the mapping between files and their constituent chunks.
|
|
// Large files are split into multiple chunks for efficient deduplication and storage.
|
|
// The Idx field maintains the order of chunks within a file.
|
|
type FileChunk struct {
|
|
FileID types.FileID
|
|
Idx int
|
|
ChunkHash types.ChunkHash
|
|
}
|
|
|
|
// Chunk represents a data chunk in the deduplication system.
|
|
// Files are split into chunks which are content-addressed by their hash.
|
|
// The ChunkHash is the SHA256 hash of the chunk content, used for deduplication.
|
|
type Chunk struct {
|
|
ChunkHash types.ChunkHash
|
|
Size int64
|
|
}
|
|
|
|
// Blob represents a blob record in the database.
|
|
// A blob is Vaultik's final storage unit - a large file (up to 10GB) containing
|
|
// many compressed and encrypted chunks from multiple source files.
|
|
// Blobs are content-addressed: the filename in S3 is
|
|
// hex(SHA256(SHA256(uncompressed blob contents))), computed from the chunk data
|
|
// before compression and encryption (not from the stored bytes). See
|
|
// blobgen.DoubleSHA256 and docs/REPOSTRUCTURE.md.
|
|
type Blob struct {
|
|
ID types.BlobID // UUID assigned when blob creation starts
|
|
|
|
// Hash is hex(SHA256(SHA256(uncompressed blob contents)))
|
|
// (empty until finalized); see the type comment above.
|
|
Hash types.BlobHash
|
|
CreatedTS time.Time // When blob creation started
|
|
FinishedTS *time.Time // When blob was finalized (nil if still packing)
|
|
UncompressedSize int64 // Total size of raw chunks before compression
|
|
CompressedSize int64 // Size after compression and encryption
|
|
UploadedTS *time.Time // When blob was uploaded to S3 (nil if not uploaded)
|
|
}
|
|
|
|
// BlobChunk represents the mapping between blobs and the chunks they contain.
|
|
// This allows tracking which chunks are stored in which blobs, along with
|
|
// their position and size within the blob. The offset and length fields
|
|
// enable extracting specific chunks from a blob without processing the entire blob.
|
|
type BlobChunk struct {
|
|
BlobID types.BlobID
|
|
ChunkHash types.ChunkHash
|
|
Offset int64
|
|
Length int64
|
|
}
|
|
|
|
// ChunkFile represents the reverse mapping showing which files contain a
|
|
// specific chunk. This is used during deduplication to identify all files
|
|
// that share a chunk, which is important for garbage collection and
|
|
// integrity verification.
|
|
type ChunkFile struct {
|
|
ChunkHash types.ChunkHash
|
|
FileID types.FileID
|
|
FileOffset int64
|
|
Length int64
|
|
}
|
|
|
|
// Snapshot represents a snapshot record in the database
|
|
type Snapshot struct {
|
|
ID types.SnapshotID
|
|
Hostname types.Hostname
|
|
VaultikVersion types.Version
|
|
VaultikGitRevision types.GitRevision
|
|
StartedAt time.Time
|
|
CompletedAt *time.Time // nil if still in progress
|
|
FileCount int64
|
|
ChunkCount int64
|
|
BlobCount int64
|
|
TotalSize int64 // Total size of all referenced files
|
|
|
|
// BlobSize is the total size of all referenced blobs (compressed and
|
|
// encrypted).
|
|
BlobSize int64
|
|
BlobUncompressedSize int64 // Total uncompressed size of all referenced blobs
|
|
CompressionRatio float64 // Compression ratio (BlobSize / BlobUncompressedSize)
|
|
CompressionLevel int // Compression level used for this snapshot
|
|
UploadBytes int64 // Total bytes uploaded during this snapshot
|
|
UploadDurationMs int64 // Total milliseconds spent uploading to S3
|
|
}
|
|
|
|
// IsComplete returns true if the snapshot has completed
|
|
func (s *Snapshot) IsComplete() bool {
|
|
return s.CompletedAt != nil
|
|
}
|
|
|
|
// SnapshotFile represents the mapping between snapshots and files
|
|
type SnapshotFile struct {
|
|
SnapshotID types.SnapshotID
|
|
FileID types.FileID
|
|
}
|
|
|
|
// SnapshotBlob represents the mapping between snapshots and blobs
|
|
type SnapshotBlob struct {
|
|
SnapshotID types.SnapshotID
|
|
BlobID types.BlobID
|
|
BlobHash types.BlobHash // Denormalized for easier manifest generation
|
|
}
|