Files
quak/src/backup.ts
T
clawbot 10e1a9ef39
check / check (push) Successful in 1m25s
Save path ./photos/YYYY/YYYY-MM/YYYY-MM-DD/YYYY-MM-DD.fileID.ext; download() from the cache first (closes #143)
Originals are saved at `{downloadDirectory}/YYYY/YYYY-MM/YYYY-MM-DD/YYYY-MM-DD.{fileID}{ext}`. The date is the photo's `takenAt` in local time, and `downloadDirectory` defaults to `./photos`, resolved when the library opens. The old `originals/` layout is gone.

`photo.download()` writes the original to `savePath`. It copies from the cache when the cache holds the original, and fetches otherwise. `lib.backup()` uses the same path and rule, and every album in `collections/` links to it. `isLocal` is true only when the original is at `savePath`.

For a file in several albums, one rule picks the copy everything uses: the most recently synced, with the lowest album ID breaking a tie.

Model: opus-5-5
2026-10-01 23:20:58 +02:00

606 lines
22 KiB
TypeScript

// The backup command, rebuilt on the library API (issue #51).
//
// `lib.backup()` waits for a completed refresh of the library (a failed one
// fails the backup before any file is touched), then, for every file in scope,
// puts its original at its save path under `downloadDirectory`, as
// `Photo.download()` does, and rebuilds the derived views (per-file sidecars,
// per-collection symlink trees, per-collection JSON) from the model. The
// on-disk layout:
//
// <downloadDirectory>/
// YYYY/YYYY-MM/YYYY-MM-DD/
// YYYY-MM-DD.<fileID>.<ext> the decrypted bytes (the save path)
// YYYY-MM-DD.<fileID>.json per-file metadata sidecar
// collections/<name>/<title> symlink to the original
// collections/<name>.json per-collection metadata
// failures.json durable ledger of unresolved failures
//
// A live photo's original is its image and its video, each with its own
// extension, beside `YYYY-MM-DD.<fileID>.livephoto.json` naming them; its album
// folders link both.
//
// Crash-safety rests on two properties. Bytes are present-means-complete: an
// original appears at its save path only via the content layer's atomic
// temp-then-rename, so a file that exists is whole and is never re-fetched — an
// interrupted run resumes by looking at the save paths. The derived views hold
// no unique state, so they are rebuilt every run; that repairs stale sidecars
// and missing or broken symlinks left by an earlier crash. A rebuild also
// removes the symlinks to originals that no longer belong to an album, and the
// directories of albums that no longer exist.
//
// Resilience (issue #8): no per-file condition aborts the run. A failed
// download or a failed symlink is caught, recorded in `failures.json` with a
// classification, a running attempt count, and the last-tried time, and the run
// continues. `result.failed` — and thus the CLI's exit code — stays non-zero
// while any failure remains unresolved and clears once every one succeeds. Each
// run reconciles the ledger against the files it attempted, so an entry for a
// file that has since left the library (deleted) or this run's scope is dropped
// rather than counted forever, which would poison a scheduled backup's exit code.
import {
lstatSync,
mkdirSync,
readdirSync,
readFileSync,
readlinkSync,
rmdirSync,
rmSync,
statSync,
symlinkSync,
writeFileSync,
} from "node:fs";
import { dirname, extname, join, relative, resolve } from "node:path";
import { removeLeftoverTempFiles } from "./download/index.js";
import { sanitizeFileName, withExtension } from "./filename.js";
import {
copyAtomic,
placeOriginal,
savePath,
storedAtSavePath,
} from "./library/content.js";
import { representative } from "./library/records.js";
import type { Collection, EnteFile } from "./model/types.js";
export type ProgressCallback = (message: string) => void;
export interface BackupOptions {
// Where the backup tree lives. `lib.backup()` defaults it to the library's
// download directory; `runBackup` with none throws before any network
// traffic.
downloadDirectory?: string;
// Fetch and store full-resolution originals. Default true.
includeOriginals?: boolean;
// Also fetch and store thumbnails under `thumbnails/<fileID>.jpg`. Default
// false.
includeThumbnails?: boolean;
// Restrict the backup to albums with these names; others are left untouched.
onlyAlbumNames?: string[];
onProgress?: ProgressCallback;
}
export interface BackupError {
fileID: number;
title: string;
collection: string;
error: string;
}
export interface BackupResult {
// Distinct files in scope this run.
totalFiles: number;
// Originals fetched (or copied from the cache) this run.
downloaded: number;
// Originals already at their save path and left untouched.
skipped: number;
// Files with an unresolved failure after this run (the ledger size); the
// CLI exits non-zero while this is above zero. A file can be both
// downloaded and failed if its bytes landed but its symlink did not.
failed: number;
// This run's per-file errors, in encounter order.
errors: BackupError[];
}
// The slice of the library that backup drives. `Library` implements it; a test
// can drive backup with a stand-in.
export interface BackupLibrary {
refresh(): Promise<void>;
listCollections(): Collection[];
listFiles(collectionID: number): EnteFile[];
// Get an original's bytes onto disk through the content cache/pools,
// returning where they landed: `destination` when they were fetched now,
// otherwise wherever they already were (the cache, or the library's save
// path). A live photo lands as its image and its video, fetched now beside
// `destination`.
original(
fileID: number,
destination: string,
): Promise<{ path: string; videoPath?: string }>;
thumbnail(fileID: number): Promise<{ path: string }>;
}
type FailureClass = "transient" | "permanent" | "unknown";
interface FailureEntry {
fileID: number;
title: string;
classification: FailureClass;
attempts: number;
lastTriedAt: number;
error: string;
}
const LEDGER_VERSION = 1;
// A regular file with content is treated as complete. A zero-byte file is not:
// it is the shape an aborted write leaves and must be re-fetched.
const isPresent = (path: string): boolean => {
try {
const s = statSync(path);
return s.isFile() && s.size > 0;
} catch {
return false;
}
};
// Best-effort classification for the ledger. Retryable server/network problems
// are transient; refusals and local filesystem/decrypt errors are permanent;
// anything else is unknown. Both the error code and message are inspected.
const classify = (err: unknown): FailureClass => {
const e = err as NodeJS.ErrnoException;
const text =
`${e?.code ?? ""} ${err instanceof Error ? err.message : String(err)}`.toLowerCase();
if (
/timeout|timed out|econnreset|econnrefused|econnaborted|network|socket|eai_again|throttl|temporarily|429|500|502|503|504/.test(
text,
)
) {
return "transient";
}
if (
/enoent|eacces|eperm|eexist|eisdir|enotempty|erofs|enospc|not found|forbidden|unauthor|decrypt|truncat|401|403|404/.test(
text,
)
) {
return "permanent";
}
return "unknown";
};
const errorMessage = (err: unknown): string =>
err instanceof Error ? err.message : String(err);
// Ensure `linkPath` is a symlink to `target`, rebuilding a missing, wrong, or
// non-symlink entry. Throws on failure (a directory in the way, no permission)
// so the caller records it and moves on rather than aborting the run.
const rebuildSymlink = (linkPath: string, target: string): void => {
try {
const st = lstatSync(linkPath);
if (st.isSymbolicLink() && readlinkSync(linkPath) === target) return;
} catch {
// Nothing there (or unreadable): fall through to create it.
}
// Remove a wrong symlink or stray file. `force` ignores a missing path but
// still refuses a directory (no `recursive`), which surfaces as a failure.
rmSync(linkPath, { force: true });
symlinkSync(target, linkPath);
};
// The on-disk names for the entries of one directory, in entry order. Each
// name is used as is unless another entry would get the same name, ignoring
// case (two names that differ only in case are one entry on a case-insensitive
// file system); then every entry sharing it gets ` (<id>)`, before the
// extension when `beforeExtension` is set. A name with an ID added can match
// another entry's own name (`IMG (6).JPG`), so this repeats until no name is
// shared. IDs are stable, so the names are too.
const uniqueNames = (
entries: { id: number; name: string }[],
beforeExtension: boolean,
): string[] => {
const withID = (id: number, name: string): string => {
const ext = beforeExtension ? extname(name) : "";
const stem = name.slice(0, name.length - ext.length);
return `${stem} (${id})${ext}`;
};
const names = entries.map((e) => e.name);
const suffixed = new Set<number>();
for (;;) {
const counts = new Map<string, number>();
for (const name of names) {
const key = name.toLowerCase();
counts.set(key, (counts.get(key) ?? 0) + 1);
}
let changed = false;
for (const [i, { id, name }] of entries.entries()) {
if (suffixed.has(i)) continue;
if (counts.get(name.toLowerCase()) === 1) continue;
names[i] = withID(id, name);
suffixed.add(i);
changed = true;
}
if (!changed) return names;
}
};
// The links a file gets in its album's folder: one named after its title, to
// its original if that is stored. A stored live photo gets two, to its image
// and its video, each named after the title with that file's extension.
const linksFor = (
file: EnteFile,
stored: { path: string; videoPath?: string } | undefined,
): { id: number; name: string; file: EnteFile; target?: string }[] => {
const name = sanitizeFileName(file.metadata.title, `file-${file.id}`);
if (stored?.videoPath === undefined) {
return [{ id: file.id, name, file, target: stored?.path }];
}
return [stored.path, stored.videoPath].map((target) => ({
id: file.id,
name: withExtension(name, extname(target)),
file,
target,
}));
};
// Every date folder (`YYYY/YYYY-MM/YYYY-MM-DD/`) under `root`, whether or not a
// file in this backup is saved there. A folder that cannot be read is skipped.
const dateFolders = (root: string): string[] => {
const subfolders = (dir: string, name: RegExp): string[] => {
try {
return readdirSync(dir, { withFileTypes: true })
.filter((e) => e.isDirectory() && name.test(e.name))
.map((e) => join(dir, e.name));
} catch {
return [];
}
};
return subfolders(root, /^\d{4}$/)
.flatMap((year) => subfolders(year, /^\d{4}-\d\d$/))
.flatMap((month) => subfolders(month, /^\d{4}-\d\d-\d\d$/));
};
// Whether the entry at `path` is a symlink a backup to `root` made: one to an
// original in a `YYYY/YYYY-MM/YYYY-MM-DD/` folder of `root`.
const linksToOriginal = (path: string, root: string): boolean => {
if (!lstatSync(path).isSymbolicLink()) return false;
const target = relative(root, resolve(dirname(path), readlinkSync(path)));
return /^\d{4}\/\d{4}-\d\d\/\d{4}-\d\d-\d\d\/[^/]+$/.test(target);
};
// Remove the symlinks in the album directory `dir` that point to an original
// in `root` and are not named in `keep`. Nothing else in the directory is
// touched: anything else there was put there by the user.
const removeStaleLinks = (
dir: string,
keep: Set<string>,
root: string,
): void => {
for (const name of readdirSync(dir)) {
if (keep.has(name)) continue;
const path = join(dir, name);
if (linksToOriginal(path, root)) rmSync(path);
}
};
// Remove the directories under `collectionsDir` that an earlier run wrote for
// an album that is gone or renamed: a directory not named in `current` with a
// `<name>.json` beside it holding an album ID, which is what a run writes. Its
// symlinks to originals in `root` are removed; if that leaves it empty, it and
// its JSON are deleted, otherwise both stay for what the user put there.
const removeStaleAlbumDirs = (
collectionsDir: string,
current: Set<string>,
root: string,
): void => {
for (const entry of readdirSync(collectionsDir, { withFileTypes: true })) {
if (!entry.isDirectory() || current.has(entry.name)) continue;
const jsonPath = join(collectionsDir, `${entry.name}.json`);
try {
const album = JSON.parse(readFileSync(jsonPath, "utf-8")) as {
id?: unknown;
};
if (typeof album.id !== "number") continue;
} catch {
continue;
}
const dir = join(collectionsDir, entry.name);
removeStaleLinks(dir, new Set(), root);
if (readdirSync(dir).length > 0) continue;
rmdirSync(dir);
rmSync(jsonPath);
}
};
const loadLedger = (path: string): Map<number, FailureEntry> => {
const ledger = new Map<number, FailureEntry>();
try {
const parsed = JSON.parse(readFileSync(path, "utf-8")) as {
files?: Record<string, FailureEntry>;
};
for (const entry of Object.values(parsed.files ?? {})) {
if (entry && typeof entry.fileID === "number") {
ledger.set(entry.fileID, entry);
}
}
} catch {
// No ledger yet, or an unreadable one: start clean.
}
return ledger;
};
const saveLedger = (path: string, ledger: Map<number, FailureEntry>): void => {
if (ledger.size === 0) {
rmSync(path, { force: true });
return;
}
const files: Record<string, FailureEntry> = {};
for (const [fileID, entry] of ledger) files[String(fileID)] = entry;
writeFileSync(
path,
JSON.stringify({ version: LEDGER_VERSION, files }, null, 2),
);
};
const writeSidecar = (path: string, file: EnteFile): void => {
const meta: Record<string, unknown> = {
id: file.id,
collectionID: file.collectionID,
ownerID: file.ownerID,
metadata: file.metadata,
};
if (file.magicMetadata) meta.magicMetadata = file.magicMetadata;
if (file.pubMagicMetadata) meta.pubMagicMetadata = file.pubMagicMetadata;
writeFileSync(path, JSON.stringify(meta, null, 2));
};
export const runBackup = async (
lib: BackupLibrary,
opts: BackupOptions,
): Promise<BackupResult> => {
const downloadDirectory = opts.downloadDirectory;
if (!downloadDirectory) {
throw new Error(
"backup requires a downloadDirectory (pass one to backup() or " +
"open the library with one)",
);
}
const includeOriginals = opts.includeOriginals ?? true;
const includeThumbnails = opts.includeThumbnails ?? false;
const log = opts.onProgress ?? (() => {});
const only = opts.onlyAlbumNames ? new Set(opts.onlyAlbumNames) : undefined;
log("Refreshing library...");
await lib.refresh();
const collectionsDir = join(downloadDirectory, "collections");
const thumbnailsDir = join(downloadDirectory, "thumbnails");
mkdirSync(collectionsDir, { recursive: true });
if (includeThumbnails) mkdirSync(thumbnailsDir, { recursive: true });
removeLeftoverTempFiles(thumbnailsDir);
for (const dir of dateFolders(downloadDirectory)) {
removeLeftoverTempFiles(dir);
}
const ledgerPath = join(downloadDirectory, "failures.json");
const ledger = loadLedger(ledgerPath);
const now = Date.now();
// Collections in scope, and the distinct files across them (a file shared
// by two albums is one original). Each file is the membership
// `representative` picks from all of its albums, in scope or not, so it is
// saved at the path `photo.savePath` names.
const allCollections = lib.listCollections();
const collections = allCollections.filter((c) =>
only ? only.has(c.name) : true,
);
const collectionName = new Map<number, string>();
for (const c of allCollections) collectionName.set(c.id, c.name);
const memberships = new Map<number, EnteFile[]>();
const filesByCollection = new Map<number, EnteFile[]>();
for (const c of allCollections) {
const files = lib.listFiles(c.id);
filesByCollection.set(c.id, files);
for (const f of files) {
const arr = memberships.get(f.id);
if (arr) arr.push(f);
else memberships.set(f.id, [f]);
}
}
const distinct = new Map<number, EnteFile>();
for (const c of collections) {
for (const f of filesByCollection.get(c.id)!) {
if (!distinct.has(f.id)) {
distinct.set(f.id, representative(memberships.get(f.id)!));
}
}
}
const errors: BackupError[] = [];
const failedThisRun = new Set<number>();
let downloaded = 0;
let skipped = 0;
const recordFailure = (
file: EnteFile,
collection: string,
err: unknown,
): void => {
// Count at most one attempt per file per run: a file whose original
// and thumbnail both fail this run must not double its attempt count
// or appear twice in errors.
if (failedThisRun.has(file.id)) return;
const error = errorMessage(err);
errors.push({
fileID: file.id,
title: file.metadata.title,
collection,
error,
});
const prior = ledger.get(file.id);
ledger.set(file.id, {
fileID: file.id,
title: file.metadata.title,
classification: classify(err),
attempts: (prior?.attempts ?? 0) + 1,
lastTriedAt: now,
error,
});
failedThisRun.add(file.id);
};
// Phase 1: get the bytes. Put each pending original at its save path
// through the content cache/pools, as `Photo.download()` does, and fetch
// the optional thumbnails; a present file is left as is.
if (includeOriginals) {
for (const [fileID, file] of distinct) {
if (storedAtSavePath(downloadDirectory, file) !== undefined) {
skipped++;
continue;
}
try {
log(`Fetching original ${file.metadata.title} (${fileID})...`);
// A fetched original is written straight to its save path (a
// live photo beside it); only one that was already cached
// elsewhere is copied.
await placeOriginal(downloadDirectory, file, (dest) =>
lib.original(fileID, dest),
);
downloaded++;
} catch (err) {
log(
`FAILED original ${file.metadata.title}: ${errorMessage(err)}`,
);
recordFailure(
file,
collectionName.get(file.collectionID) ?? "",
err,
);
}
}
}
if (includeThumbnails) {
for (const [fileID, file] of distinct) {
const dest = join(thumbnailsDir, `${fileID}.jpg`);
if (isPresent(dest)) continue;
try {
const { path } = await lib.thumbnail(fileID);
await copyAtomic(path, dest);
} catch (err) {
recordFailure(
file,
collectionName.get(file.collectionID) ?? "",
err,
);
}
}
}
// Phase 2: rebuild the derived views from the model. Sidecars first, for
// every present original (this repairs stale ones).
if (includeOriginals) {
for (const file of distinct.values()) {
if (storedAtSavePath(downloadDirectory, file) !== undefined) {
const path = savePath(downloadDirectory, file);
writeSidecar(withExtension(path, ".json"), file);
}
}
}
// Then the per-collection symlink trees and JSON. Directory names are
// chosen across every album, not just those in scope, so a scoped run
// names an album the same as a full one and never takes the directory of
// an album it skipped. Stale entries are removed before anything is
// rebuilt, so on a case-insensitive file system removing an old name can
// never remove the new one.
const dirNames = uniqueNames(
allCollections.map((c) => ({
id: c.id,
name: sanitizeFileName(c.name, `collection-${c.id}`),
})),
false,
);
const albumDirNames = new Map(
allCollections.map((c, i) => [c.id, dirNames[i]!]),
);
try {
removeStaleAlbumDirs(
collectionsDir,
new Set(dirNames),
downloadDirectory,
);
} catch (err) {
log(`FAILED removing old album directories: ${errorMessage(err)}`);
}
for (const c of collections) {
const colDirName = albumDirNames.get(c.id)!;
const colDir = join(collectionsDir, colDirName);
mkdirSync(colDir, { recursive: true });
// Every album links the one original, saved from the file's entry in
// `distinct`.
const files = filesByCollection.get(c.id) ?? [];
const links = files.flatMap((f) =>
linksFor(
f,
storedAtSavePath(downloadDirectory, distinct.get(f.id)!),
),
);
const linkNames = uniqueNames(links, true);
try {
removeStaleLinks(colDir, new Set(linkNames), downloadDirectory);
} catch (err) {
log(`FAILED removing old links in ${c.name}: ${errorMessage(err)}`);
}
const metaFiles = files.map((f) => ({
id: f.id,
metadata: f.metadata,
}));
for (const [i, link] of links.entries()) {
if (!includeOriginals || link.target === undefined) continue;
const linkName = linkNames[i]!;
try {
rebuildSymlink(
join(colDir, linkName),
relative(colDir, link.target),
);
} catch (err) {
log(
`FAILED symlink ${c.name}/${linkName}: ${errorMessage(err)}`,
);
recordFailure(link.file, c.name, err);
}
}
writeFileSync(
join(collectionsDir, `${colDirName}.json`),
JSON.stringify(
{ id: c.id, name: c.name, type: c.type, files: metaFiles },
null,
2,
),
);
}
// Reconcile the ledger against what this run actually attempted: an entry
// survives only for a file that failed this run. A file that succeeded had
// its failure resolved; a file gone from the library (deleted) or outside
// this run's scope is not something this run can resolve, so keeping its
// stale entry would keep the exit code non-zero forever — a single
// since-deleted photo would fail every future scheduled backup.
for (const fileID of [...ledger.keys()]) {
if (!failedThisRun.has(fileID)) ledger.delete(fileID);
}
saveLedger(ledgerPath, ledger);
return {
totalFiles: distinct.size,
downloaded,
skipped,
failed: ledger.size,
errors,
};
};