Files
quak/src/backup.ts
T
clawbot 17d1d74615
check / check (push) Successful in 25s
Rewrite the backup command on the library API with a durable failure ledger (closes #51)
Rewrites backup on the library API. backup() refreshes, fetches pending originals (and optionally thumbnails) through the pools reusing the content cache, then materialises the unchanged collections/ symlink views + per-collection JSON + sidecars from the model. A symlink failure no longer aborts the run (closes #8). failures.json reconciles against each run's attempted set — since-deleted/out-of-scope/resolved entries clear, still-failing retained, one attempt per file per run — and the exit-code contract is preserved.

Model: opus-4-8
2026-09-22 20:21:49 +02:00

427 lines
15 KiB
TypeScript

// The backup command, rebuilt on the library API (issue #51).
//
// `lib.backup()` refreshes the library, then, for every file in scope, gets its
// original bytes onto disk under `downloadDirectory` and rebuilds the derived
// views (per-file sidecars, per-collection symlink trees, per-collection JSON)
// from the model. The on-disk layout is the historical one, unchanged:
//
// <downloadDirectory>/
// originals/<fileID>.<ext> the decrypted bytes
// originals/<fileID>.json per-file metadata sidecar
// collections/<name>/<title> symlink into ../../originals
// collections/<name>.json per-collection metadata
// failures.json durable ledger of unresolved failures
//
// Crash-safety rests on two properties. Bytes are present-means-complete: an
// original appears under `originals/` only via the content layer's atomic
// temp-then-rename, so a file that exists is whole and is never re-fetched — an
// interrupted run resumes by listing the directory. The derived views hold no
// unique state, so they are rebuilt every run; that repairs stale sidecars and
// missing or broken symlinks left by an earlier crash.
//
// Resilience (issue #8): no per-file condition aborts the run. A failed
// download or a failed symlink is caught, recorded in `failures.json` with a
// classification, a running attempt count, and the last-tried time, and the run
// continues. `result.failed` — and thus the CLI's exit code — stays non-zero
// while any failure remains unresolved and clears once every one succeeds. Each
// run reconciles the ledger against the files it attempted, so an entry for a
// file that has since left the library (deleted) or this run's scope is dropped
// rather than counted forever, which would poison a scheduled backup's exit code.
import {
copyFileSync,
lstatSync,
mkdirSync,
readFileSync,
readlinkSync,
renameSync,
rmSync,
statSync,
symlinkSync,
writeFileSync,
} from "node:fs";
import { basename, dirname, extname, join, relative } from "node:path";
import type { Collection, EnteFile } from "./model/types.js";
export type ProgressCallback = (message: string) => void;
export interface BackupOptions {
// Where the backup tree lives. Required: with none, `backup()` throws
// before any network traffic. A library opened with a `downloadDirectory`
// supplies the default.
downloadDirectory?: string;
// Fetch and store full-resolution originals. Default true.
includeOriginals?: boolean;
// Also fetch and store thumbnails under `thumbnails/<fileID>.jpg`. Default
// false.
includeThumbnails?: boolean;
// Restrict the backup to albums with these names; others are left untouched.
onlyAlbumNames?: string[];
onProgress?: ProgressCallback;
}
export interface BackupError {
fileID: number;
title: string;
collection: string;
error: string;
}
export interface BackupResult {
// Distinct files in scope this run.
totalFiles: number;
// Originals fetched (or copied from the cache) this run.
downloaded: number;
// Originals already present and left untouched.
skipped: number;
// Files with an unresolved failure after this run (the ledger size); the
// CLI exits non-zero while this is above zero. A file can be both
// downloaded and failed if its bytes landed but its symlink did not.
failed: number;
// This run's per-file errors, in encounter order.
errors: BackupError[];
}
// The slice of the library that backup drives. `Library` implements it; a test
// can drive backup with a stand-in.
export interface BackupLibrary {
refresh(): Promise<void>;
listCollections(): Collection[];
listFiles(collectionID: number): EnteFile[];
// Get an original's bytes onto disk through the content cache/pools,
// returning where they landed (the cache, or a prior backup).
original(fileID: number): Promise<{ path: string }>;
thumbnail(fileID: number): Promise<{ path: string }>;
}
type FailureClass = "transient" | "permanent" | "unknown";
interface FailureEntry {
fileID: number;
title: string;
classification: FailureClass;
attempts: number;
lastTriedAt: number;
error: string;
}
const LEDGER_VERSION = 1;
const sanitizePath = (name: string): string =>
name.replace(/[/\\:*?"<>|]/g, "_").replace(/^\.+/, "_");
// The originals/ filename for a file: `<id><ext>`, the extension taken from the
// title (or `.bin`). Matches the content cache's own naming so a present check
// lines up with what a fetch would write.
const originalName = (file: EnteFile): string => {
const ext = extname(file.metadata.title || "") || ".bin";
return `${file.id}${ext}`;
};
// A regular file with content is treated as complete. A zero-byte file is not:
// it is the shape an aborted write leaves and must be re-fetched.
const isPresent = (path: string): boolean => {
try {
const s = statSync(path);
return s.isFile() && s.size > 0;
} catch {
return false;
}
};
// Best-effort classification for the ledger. Retryable server/network problems
// are transient; refusals and local filesystem/decrypt errors are permanent;
// anything else is unknown. Both the error code and message are inspected.
const classify = (err: unknown): FailureClass => {
const e = err as NodeJS.ErrnoException;
const text =
`${e?.code ?? ""} ${err instanceof Error ? err.message : String(err)}`.toLowerCase();
if (
/timeout|timed out|econnreset|econnrefused|econnaborted|network|socket|eai_again|throttl|temporarily|429|500|502|503|504/.test(
text,
)
) {
return "transient";
}
if (
/enoent|eacces|eperm|eexist|eisdir|enotempty|erofs|enospc|not found|forbidden|unauthor|decrypt|truncat|401|403|404/.test(
text,
)
) {
return "permanent";
}
return "unknown";
};
const errorMessage = (err: unknown): string =>
err instanceof Error ? err.message : String(err);
// Copy bytes into `dest` via a temp file in the same directory plus rename, so
// `dest` appears only once it is whole ("present means complete").
const copyAtomic = (src: string, dest: string): void => {
if (src === dest) return;
const tmp = join(
dirname(dest),
`.quak-backup-${basename(dest)}-${process.pid}-${Math.random()
.toString(36)
.slice(2)}.tmp`,
);
try {
copyFileSync(src, tmp);
renameSync(tmp, dest);
} finally {
rmSync(tmp, { force: true });
}
};
// Ensure `linkPath` is a symlink to `target`, rebuilding a missing, wrong, or
// non-symlink entry. Throws on failure (a directory in the way, no permission)
// so the caller records it and moves on rather than aborting the run.
const rebuildSymlink = (linkPath: string, target: string): void => {
try {
const st = lstatSync(linkPath);
if (st.isSymbolicLink() && readlinkSync(linkPath) === target) return;
} catch {
// Nothing there (or unreadable): fall through to create it.
}
// Remove a wrong symlink or stray file. `force` ignores a missing path but
// still refuses a directory (no `recursive`), which surfaces as a failure.
rmSync(linkPath, { force: true });
symlinkSync(target, linkPath);
};
const loadLedger = (path: string): Map<number, FailureEntry> => {
const ledger = new Map<number, FailureEntry>();
try {
const parsed = JSON.parse(readFileSync(path, "utf-8")) as {
files?: Record<string, FailureEntry>;
};
for (const entry of Object.values(parsed.files ?? {})) {
if (entry && typeof entry.fileID === "number") {
ledger.set(entry.fileID, entry);
}
}
} catch {
// No ledger yet, or an unreadable one: start clean.
}
return ledger;
};
const saveLedger = (path: string, ledger: Map<number, FailureEntry>): void => {
if (ledger.size === 0) {
rmSync(path, { force: true });
return;
}
const files: Record<string, FailureEntry> = {};
for (const [fileID, entry] of ledger) files[String(fileID)] = entry;
writeFileSync(
path,
JSON.stringify({ version: LEDGER_VERSION, files }, null, 2),
);
};
const writeSidecar = (path: string, file: EnteFile): void => {
const meta: Record<string, unknown> = {
id: file.id,
collectionID: file.collectionID,
ownerID: file.ownerID,
metadata: file.metadata,
};
if (file.magicMetadata) meta.magicMetadata = file.magicMetadata;
if (file.pubMagicMetadata) meta.pubMagicMetadata = file.pubMagicMetadata;
writeFileSync(path, JSON.stringify(meta, null, 2));
};
export const runBackup = async (
lib: BackupLibrary,
opts: BackupOptions,
): Promise<BackupResult> => {
const downloadDirectory = opts.downloadDirectory;
if (!downloadDirectory) {
throw new Error(
"backup requires a downloadDirectory (pass one to backup() or " +
"open the library with one)",
);
}
const includeOriginals = opts.includeOriginals ?? true;
const includeThumbnails = opts.includeThumbnails ?? false;
const log = opts.onProgress ?? (() => {});
const only = opts.onlyAlbumNames ? new Set(opts.onlyAlbumNames) : undefined;
log("Refreshing library...");
await lib.refresh();
const originalsDir = join(downloadDirectory, "originals");
const collectionsDir = join(downloadDirectory, "collections");
const thumbnailsDir = join(downloadDirectory, "thumbnails");
mkdirSync(originalsDir, { recursive: true });
mkdirSync(collectionsDir, { recursive: true });
if (includeThumbnails) mkdirSync(thumbnailsDir, { recursive: true });
const ledgerPath = join(downloadDirectory, "failures.json");
const ledger = loadLedger(ledgerPath);
const now = Date.now();
// Collections in scope, and the distinct files across them (a file shared
// by two albums is one original).
const collections = lib
.listCollections()
.filter((c) => (only ? only.has(c.name) : true));
const collectionName = new Map<number, string>();
for (const c of collections) collectionName.set(c.id, c.name);
const distinct = new Map<number, EnteFile>();
const filesByCollection = new Map<number, EnteFile[]>();
for (const c of collections) {
const files = lib.listFiles(c.id);
filesByCollection.set(c.id, files);
for (const f of files) if (!distinct.has(f.id)) distinct.set(f.id, f);
}
const errors: BackupError[] = [];
const failedThisRun = new Set<number>();
let downloaded = 0;
let skipped = 0;
const recordFailure = (
file: EnteFile,
collection: string,
err: unknown,
): void => {
// Count at most one attempt per file per run: a file whose original
// and thumbnail both fail this run must not double its attempt count
// or appear twice in errors.
if (failedThisRun.has(file.id)) return;
const error = errorMessage(err);
errors.push({
fileID: file.id,
title: file.metadata.title,
collection,
error,
});
const prior = ledger.get(file.id);
ledger.set(file.id, {
fileID: file.id,
title: file.metadata.title,
classification: classify(err),
attempts: (prior?.attempts ?? 0) + 1,
lastTriedAt: now,
error,
});
failedThisRun.add(file.id);
};
// Phase 1: get the bytes. Fetch each pending original (and optional
// thumbnail) through the content cache/pools and place it under the backup
// tree; a present file is left as is.
if (includeOriginals) {
for (const [fileID, file] of distinct) {
const dest = join(originalsDir, originalName(file));
if (isPresent(dest)) {
skipped++;
continue;
}
try {
log(`Fetching original ${file.metadata.title} (${fileID})...`);
const { path } = await lib.original(fileID);
copyAtomic(path, dest);
downloaded++;
} catch (err) {
log(
`FAILED original ${file.metadata.title}: ${errorMessage(err)}`,
);
recordFailure(
file,
collectionName.get(file.collectionID) ?? "",
err,
);
}
}
}
if (includeThumbnails) {
for (const [fileID, file] of distinct) {
const dest = join(thumbnailsDir, `${fileID}.jpg`);
if (isPresent(dest)) continue;
try {
const { path } = await lib.thumbnail(fileID);
copyAtomic(path, dest);
} catch (err) {
recordFailure(
file,
collectionName.get(file.collectionID) ?? "",
err,
);
}
}
}
// Phase 2: rebuild the derived views from the model. Sidecars first, for
// every present original (this repairs stale ones).
if (includeOriginals) {
for (const [fileID, file] of distinct) {
const orig = join(originalsDir, originalName(file));
if (isPresent(orig)) {
writeSidecar(join(originalsDir, `${fileID}.json`), file);
}
}
}
// Then the per-collection symlink trees and JSON.
for (const c of collections) {
const colDirName = sanitizePath(c.name || `collection-${c.id}`);
const colDir = join(collectionsDir, colDirName);
mkdirSync(colDir, { recursive: true });
const files = filesByCollection.get(c.id) ?? [];
const metaFiles: { id: number; metadata: EnteFile["metadata"] }[] = [];
for (const file of files) {
metaFiles.push({ id: file.id, metadata: file.metadata });
if (!includeOriginals) continue;
const orig = join(originalsDir, originalName(file));
if (!isPresent(orig)) continue;
const linkName = sanitizePath(
file.metadata.title || `file-${file.id}`,
);
const linkPath = join(colDir, linkName);
try {
rebuildSymlink(linkPath, relative(colDir, orig));
} catch (err) {
log(
`FAILED symlink ${c.name}/${linkName}: ${errorMessage(err)}`,
);
recordFailure(file, c.name, err);
}
}
writeFileSync(
join(collectionsDir, `${colDirName}.json`),
JSON.stringify(
{ id: c.id, name: c.name, type: c.type, files: metaFiles },
null,
2,
),
);
}
// Reconcile the ledger against what this run actually attempted: an entry
// survives only for a file that failed this run. A file that succeeded had
// its failure resolved; a file gone from the library (deleted) or outside
// this run's scope is not something this run can resolve, so keeping its
// stale entry would keep the exit code non-zero forever — a single
// since-deleted photo would fail every future scheduled backup.
for (const fileID of [...ledger.keys()]) {
if (!failedThisRun.has(fileID)) ledger.delete(fileID);
}
saveLedger(ledgerPath, ledger);
return {
totalFiles: distinct.size,
downloaded,
skipped,
failed: ledger.size,
errors,
};
};