Move migrations to internal/db/migrations (closes #96)
check / check (push) Successful in 3m13s

REPO_POLICIES.md puts migrations in internal/db/migrations/ as
000_migration.sql and 001_schema.sql. The two files move there with
their contents unchanged. go:embed cannot reach outside its own
package, so internal/db/migrations has a small package that embeds
them, and internal/database reads them through its FS(). The database
package stays where CONVENTIONS.md puts it; moving it would change
existing test files in other packages.

The version still comes from the filename prefix, so a database that
has recorded versions 0 and 1 runs neither again. A new test applies
the migrations twice to one database file and checks that the second
run applies nothing.

Model: opus-5-5
This commit is contained in:
2026-09-29 05:00:17 +00:00
parent 1b920fe000
commit 72322a49c7
6 changed files with 91 additions and 17 deletions
+15 -17
View File
@@ -4,9 +4,9 @@ package database
import (
"context"
"database/sql"
"embed"
"errors"
"fmt"
"io/fs"
"log/slog"
"path/filepath"
"sort"
@@ -15,14 +15,12 @@ import (
"go.uber.org/fx"
"sneak.berlin/go/pixa/internal/config"
"sneak.berlin/go/pixa/internal/db/migrations"
"sneak.berlin/go/pixa/internal/logger"
_ "modernc.org/sqlite" // SQLite driver registration
)
//go:embed schema/*.sql
var schemaFS embed.FS
// bootstrapVersion is the migration that creates the schema_migrations
// table itself. It is applied before the normal migration loop.
const bootstrapVersion = 0
@@ -113,29 +111,29 @@ func New(lc fx.Lifecycle, params Params) (*Database, error) {
return s, nil
}
// collectMigrations reads the embedded schema directory and returns
// collectMigrations reads the embedded migrations directory and returns
// migration filenames sorted lexicographically.
func collectMigrations() ([]string, error) {
entries, err := schemaFS.ReadDir("schema")
entries, err := fs.ReadDir(migrations.FS(), ".")
if err != nil {
return nil, fmt.Errorf("failed to read schema directory: %w", err)
return nil, fmt.Errorf("failed to read migrations directory: %w", err)
}
var migrations []string
var filenames []string
for _, entry := range entries {
if !entry.IsDir() && strings.HasSuffix(entry.Name(), ".sql") {
migrations = append(migrations, entry.Name())
filenames = append(filenames, entry.Name())
}
}
sort.Strings(migrations)
sort.Strings(filenames)
return migrations, nil
return filenames, nil
}
// bootstrapMigrationsTable ensures the schema_migrations table exists
// by applying 000.sql if the table is missing.
// by applying 000_migration.sql if the table is missing.
func bootstrapMigrationsTable(ctx context.Context, db *sql.DB, log *slog.Logger) error {
var tableExists int
@@ -150,9 +148,9 @@ func bootstrapMigrationsTable(ctx context.Context, db *sql.DB, log *slog.Logger)
return nil
}
content, err := schemaFS.ReadFile("schema/000.sql")
content, err := fs.ReadFile(migrations.FS(), "000_migration.sql")
if err != nil {
return fmt.Errorf("failed to read bootstrap migration 000.sql: %w", err)
return fmt.Errorf("failed to read bootstrap migration 000_migration.sql: %w", err)
}
if log != nil {
@@ -177,12 +175,12 @@ func ApplyMigrations(ctx context.Context, db *sql.DB, log *slog.Logger) error {
return err
}
migrations, err := collectMigrations()
filenames, err := collectMigrations()
if err != nil {
return err
}
for _, migration := range migrations {
for _, migration := range filenames {
version, parseErr := ParseMigrationVersion(migration)
if parseErr != nil {
return parseErr
@@ -208,7 +206,7 @@ func ApplyMigrations(ctx context.Context, db *sql.DB, log *slog.Logger) error {
}
// Read and apply migration.
content, readErr := schemaFS.ReadFile(filepath.Join("schema", migration))
content, readErr := fs.ReadFile(migrations.FS(), migration)
if readErr != nil {
return fmt.Errorf("failed to read migration %s: %w", migration, readErr)
}
@@ -0,0 +1,54 @@
package database
import (
"bytes"
"database/sql"
"log/slog"
"path/filepath"
"strings"
"testing"
_ "modernc.org/sqlite" // SQLite driver registration
)
// TestApplyMigrations_SecondRunAppliesNothing applies the migrations twice
// to one database file, as happens when pixad starts again on the database
// it created, and checks that the second run applies none of them.
// ApplyMigrations logs a message starting with "applying" before it runs
// any migration, the bootstrap one included.
func TestApplyMigrations_SecondRunAppliesNothing(t *testing.T) {
t.Parallel()
ctx := t.Context()
db, err := sql.Open("sqlite", filepath.Join(t.TempDir(), "state.sqlite3"))
if err != nil {
t.Fatalf("failed to open test db: %v", err)
}
t.Cleanup(func() { _ = db.Close() })
var firstLog bytes.Buffer
err = ApplyMigrations(ctx, db, slog.New(slog.NewTextHandler(&firstLog, nil)))
if err != nil {
t.Fatalf("first ApplyMigrations failed: %v", err)
}
if !strings.Contains(firstLog.String(), "applying") {
t.Fatalf("first ApplyMigrations logged no applied migration:\n%s",
firstLog.String())
}
var secondLog bytes.Buffer
err = ApplyMigrations(ctx, db, slog.New(slog.NewTextHandler(&secondLog, nil)))
if err != nil {
t.Fatalf("second ApplyMigrations failed: %v", err)
}
if strings.Contains(secondLog.String(), "applying") {
t.Errorf("second ApplyMigrations ran a migration again:\n%s",
secondLog.String())
}
}
-9
View File
@@ -1,9 +0,0 @@
-- Migration 000: Schema migrations tracking table
-- Applied as a bootstrap step before the normal migration loop.
CREATE TABLE IF NOT EXISTS schema_migrations (
version INTEGER PRIMARY KEY,
applied_at DATETIME DEFAULT CURRENT_TIMESTAMP
);
INSERT OR IGNORE INTO schema_migrations (version) VALUES (0);
@@ -1,112 +0,0 @@
-- Migration 001: Initial schema
-- Creates all tables for the pixa caching image proxy
-- Source content blobs
-- Files stored at: cache/src-content/<ab>/<cd>/<sha256>
-- last_accessed_at is NULL until the first LRU touch; eviction falls
-- back to fetched_at for rows that have never been touched.
CREATE TABLE IF NOT EXISTS source_content (
content_hash TEXT PRIMARY KEY,
content_type TEXT NOT NULL,
size_bytes INTEGER NOT NULL,
fetched_at DATETIME DEFAULT CURRENT_TIMESTAMP,
last_accessed_at DATETIME
);
CREATE INDEX IF NOT EXISTS idx_source_content_last_accessed
ON source_content(last_accessed_at);
-- Source URL metadata - maps URLs to content hashes
-- JSON stored at: cache/src-metadata/<hostname>/<path_hash>.json
CREATE TABLE IF NOT EXISTS source_metadata (
id INTEGER PRIMARY KEY AUTOINCREMENT,
source_host TEXT NOT NULL,
source_path TEXT NOT NULL,
source_query TEXT NOT NULL DEFAULT '',
path_hash TEXT NOT NULL,
content_hash TEXT,
status_code INTEGER NOT NULL,
content_type TEXT,
response_headers TEXT,
fetched_at DATETIME DEFAULT CURRENT_TIMESTAMP,
expires_at DATETIME,
etag TEXT,
last_modified TEXT,
UNIQUE(source_host, source_path, source_query),
FOREIGN KEY (content_hash) REFERENCES source_content(content_hash)
);
CREATE INDEX IF NOT EXISTS idx_source_meta_host ON source_metadata(source_host);
CREATE INDEX IF NOT EXISTS idx_source_meta_path_hash ON source_metadata(path_hash);
CREATE INDEX IF NOT EXISTS idx_source_meta_expires ON source_metadata(expires_at);
CREATE INDEX IF NOT EXISTS idx_source_meta_content_hash ON source_metadata(content_hash);
-- Processed variant blobs
-- Files stored at: cache/variants/<ab>/<cd>/<cache_key> (plus a .meta
-- sidecar with the content type). Tracked here (like source content
-- blobs above) so total cache usage can be computed with a SUM query,
-- never a directory scan, and so LRU eviction has a timestamp to order
-- on.
CREATE TABLE IF NOT EXISTS variant_content (
cache_key TEXT PRIMARY KEY,
size_bytes INTEGER NOT NULL,
content_type TEXT NOT NULL,
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
last_accessed_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP
);
CREATE INDEX IF NOT EXISTS idx_variant_content_last_accessed
ON variant_content(last_accessed_at);
-- Output/transformed content blobs
-- Files stored at: cache/dst-content/<ab>/<cd>/<sha256>
CREATE TABLE IF NOT EXISTS output_content (
content_hash TEXT PRIMARY KEY,
content_type TEXT NOT NULL,
size_bytes INTEGER NOT NULL,
fetched_at DATETIME DEFAULT CURRENT_TIMESTAMP
);
-- Request cache - maps full request params to output content
CREATE TABLE IF NOT EXISTS request_cache (
id INTEGER PRIMARY KEY AUTOINCREMENT,
cache_key TEXT NOT NULL UNIQUE,
source_metadata_id INTEGER NOT NULL,
output_hash TEXT NOT NULL,
width INTEGER NOT NULL,
height INTEGER NOT NULL,
format TEXT NOT NULL,
quality INTEGER NOT NULL DEFAULT 85,
fit_mode TEXT NOT NULL DEFAULT 'cover',
fetched_at DATETIME DEFAULT CURRENT_TIMESTAMP,
access_count INTEGER NOT NULL DEFAULT 1,
FOREIGN KEY (source_metadata_id) REFERENCES source_metadata(id),
FOREIGN KEY (output_hash) REFERENCES output_content(content_hash)
);
CREATE INDEX IF NOT EXISTS idx_request_cache_key ON request_cache(cache_key);
CREATE INDEX IF NOT EXISTS idx_request_cache_source ON request_cache(source_metadata_id);
CREATE INDEX IF NOT EXISTS idx_request_cache_output ON request_cache(output_hash);
CREATE INDEX IF NOT EXISTS idx_request_cache_fetched ON request_cache(fetched_at);
-- Negative cache for failed fetches (404s, timeouts, etc.)
CREATE TABLE IF NOT EXISTS negative_cache (
id INTEGER PRIMARY KEY AUTOINCREMENT,
source_host TEXT NOT NULL,
source_path TEXT NOT NULL,
source_query TEXT NOT NULL DEFAULT '',
status_code INTEGER NOT NULL,
error_message TEXT,
fetched_at DATETIME DEFAULT CURRENT_TIMESTAMP,
expires_at DATETIME NOT NULL,
UNIQUE(source_host, source_path, source_query)
);
CREATE INDEX IF NOT EXISTS idx_negative_cache_expires ON negative_cache(expires_at);
-- Cache statistics for monitoring
CREATE TABLE IF NOT EXISTS cache_stats (
id INTEGER PRIMARY KEY CHECK (id = 1),
hit_count INTEGER NOT NULL DEFAULT 0,
miss_count INTEGER NOT NULL DEFAULT 0,
upstream_fetch_count INTEGER NOT NULL DEFAULT 0,
upstream_fetch_bytes INTEGER NOT NULL DEFAULT 0,
transform_count INTEGER NOT NULL DEFAULT 0,
last_updated_at DATETIME DEFAULT CURRENT_TIMESTAMP
);
INSERT OR IGNORE INTO cache_stats (id) VALUES (1);