check / check (push) Waiting to run
lib.backup() takes a lock, backup.lock in its download directory, made with proper-lockfile, before its refresh, and removes it when it ends. A second backup of the directory fails at once with an error naming it; quak backup prints it as one line and exits 2. A lock untouched for 10 seconds, left by a killed run, is taken over. Deviation: yarn.lock was regenerated by yarn add in the pinned node image. Judgement call: a run failing at its refresh leaves the directory, empty. Judgement call: a lock removed mid-run stops that run with an uncaught error, the library's default. Model: opus-5-5
773 lines
29 KiB
TypeScript
773 lines
29 KiB
TypeScript
// The backup command, rebuilt on the library API (issue #51).
|
|
//
|
|
// `lib.backup()` takes the lock in `downloadDirectory`, and fails at once when
|
|
// another backup of it holds the lock. It waits for a completed refresh of the
|
|
// library (a failed one fails the backup before any file is touched), then, for
|
|
// every file in scope, puts its original at its save path under
|
|
// `downloadDirectory`, as `Photo.download()` does, waits for an ML data fetch,
|
|
// and rebuilds the derived views (per-file sidecars, per-collection symlink
|
|
// trees, per-collection JSON) from the model. The on-disk layout:
|
|
//
|
|
// <downloadDirectory>/
|
|
// YYYY/YYYY-MM/YYYY-MM-DD/
|
|
// YYYY-MM-DD.<fileID>.<ext> the decrypted bytes (the save path)
|
|
// YYYY-MM-DD.<fileID>.json per-file metadata sidecar, with
|
|
// the file's ML data and its
|
|
// original's EXIF, XMP and
|
|
// dimensions
|
|
// collections/<name>/<title> symlink to the original
|
|
// collections/<name>.json per-collection metadata
|
|
// account.json the account's email and user ID
|
|
// backup.lock the lock, while a backup runs
|
|
// failures.json durable ledger of unresolved failures
|
|
//
|
|
// A live photo's original is its image and its video, each with its own
|
|
// extension, beside `YYYY-MM-DD.<fileID>.livephoto.json` naming them; its album
|
|
// folders link both.
|
|
//
|
|
// Crash-safety rests on two properties. Bytes are present-means-complete: an
|
|
// original appears at its save path only via the content layer's atomic
|
|
// temp-then-rename, so a file that exists is whole and is never re-fetched — an
|
|
// interrupted run resumes by looking at the save paths. The derived views hold
|
|
// no unique state, so they are rebuilt every run; that repairs stale sidecars
|
|
// and missing or broken symlinks left by an earlier crash. A rebuild also
|
|
// removes the symlinks to originals that no longer belong to an album, and the
|
|
// directories of albums that no longer exist. The one thing a sidecar takes
|
|
// from the sidecar it replaces is its original's EXIF, XMP and dimensions (or
|
|
// why they could not be read), so that a run does not read every stored
|
|
// original again; a sidecar without them gets them read from the original.
|
|
//
|
|
// Resilience (issue #8): no per-file condition aborts the run. A failed
|
|
// download, a failed symlink, or ML data missing because the ML data fetch
|
|
// failed is caught, recorded in `failures.json` with a classification, a
|
|
// running attempt count, and the last-tried time, and the run continues.
|
|
// `result.failed` — and thus the CLI's exit code — stays non-zero while any
|
|
// failure remains unresolved and clears once every one succeeds. Each run
|
|
// reconciles the ledger against the files it attempted, so an entry for a file
|
|
// that has since left the library (deleted) or this run's scope is dropped
|
|
// rather than counted forever, which would poison a scheduled backup's exit
|
|
// code.
|
|
|
|
import {
|
|
lstatSync,
|
|
mkdirSync,
|
|
readdirSync,
|
|
readFileSync,
|
|
readlinkSync,
|
|
rmdirSync,
|
|
rmSync,
|
|
statSync,
|
|
symlinkSync,
|
|
writeFileSync,
|
|
} from "node:fs";
|
|
import { readFile } from "node:fs/promises";
|
|
import { dirname, extname, join, relative, resolve } from "node:path";
|
|
import lockfile from "proper-lockfile";
|
|
|
|
import { removeLeftoverTempFiles } from "./download/index.js";
|
|
import { sanitizeFileName, withExtension } from "./filename.js";
|
|
import {
|
|
copyAtomic,
|
|
placeOriginal,
|
|
savePath,
|
|
storedAtSavePath,
|
|
} from "./library/content.js";
|
|
import { representative } from "./library/records.js";
|
|
import { extractImageMetadata } from "./metadata-backup.js";
|
|
import type { MLData } from "./mldata-fetch.js";
|
|
import type { Collection, EnteFile, FileMetadata } from "./model/types.js";
|
|
|
|
export type ProgressCallback = (message: string) => void;
|
|
|
|
export interface BackupOptions {
|
|
// Where the backup tree lives. `lib.backup()` defaults it to the library's
|
|
// download directory; `runBackup` with none throws before any network
|
|
// traffic.
|
|
downloadDirectory?: string;
|
|
// Fetch and store full-resolution originals. Default true.
|
|
includeOriginals?: boolean;
|
|
// Also fetch and store thumbnails under `thumbnails/<fileID>.jpg`. Default
|
|
// false.
|
|
includeThumbnails?: boolean;
|
|
// Restrict the backup to albums with these names; others are left untouched.
|
|
onlyAlbumNames?: string[];
|
|
onProgress?: ProgressCallback;
|
|
}
|
|
|
|
export interface BackupError {
|
|
fileID: number;
|
|
title: string;
|
|
collection: string;
|
|
error: string;
|
|
}
|
|
|
|
export interface BackupResult {
|
|
// Distinct files in scope this run.
|
|
totalFiles: number;
|
|
// Originals fetched (or copied from the cache) this run.
|
|
downloaded: number;
|
|
// Originals already at their save path and left untouched.
|
|
skipped: number;
|
|
// Files with an unresolved failure after this run (the ledger size); the
|
|
// CLI exits non-zero while this is above zero. A file can be both
|
|
// downloaded and failed if its bytes landed but its symlink did not.
|
|
failed: number;
|
|
// This run's per-file errors, in encounter order.
|
|
errors: BackupError[];
|
|
}
|
|
|
|
// The slice of the library that backup drives. `Library` implements it; a test
|
|
// can drive backup with a stand-in.
|
|
export interface BackupLibrary {
|
|
// The account the library belongs to.
|
|
whoami(): { email: string; userID: number };
|
|
refresh(): Promise<void>;
|
|
listCollections(): Collection[];
|
|
listFiles(collectionID: number): EnteFile[];
|
|
// Get an original's bytes onto disk through the content cache/pools,
|
|
// returning where they landed: `destination` when they were fetched now,
|
|
// otherwise wherever they already were (the cache, or the library's save
|
|
// path). A live photo lands as its image and its video, fetched now beside
|
|
// `destination`.
|
|
original(
|
|
fileID: number,
|
|
destination: string,
|
|
): Promise<{ path: string; videoPath?: string }>;
|
|
thumbnail(fileID: number): Promise<{ path: string }>;
|
|
// Wait for an ML data fetch to complete, joining one already running or
|
|
// starting one. Rejects with the reason when it fails; resolves at once
|
|
// when the library cannot fetch ML data.
|
|
fetchMLData(): Promise<void>;
|
|
// A file's cached ML data, as `lib.mldata.forFile()` returns it, or
|
|
// undefined when none is cached.
|
|
mlData(fileID: number): Promise<MLData | undefined>;
|
|
}
|
|
|
|
type FailureClass = "transient" | "permanent" | "unknown";
|
|
|
|
interface FailureEntry {
|
|
fileID: number;
|
|
title: string;
|
|
classification: FailureClass;
|
|
attempts: number;
|
|
lastTriedAt: number;
|
|
error: string;
|
|
}
|
|
|
|
const LEDGER_VERSION = 1;
|
|
|
|
// A regular file with content is treated as complete. A zero-byte file is not:
|
|
// it is the shape an aborted write leaves and must be re-fetched.
|
|
const isPresent = (path: string): boolean => {
|
|
try {
|
|
const s = statSync(path);
|
|
return s.isFile() && s.size > 0;
|
|
} catch {
|
|
return false;
|
|
}
|
|
};
|
|
|
|
// Best-effort classification for the ledger. Retryable server/network problems
|
|
// are transient; refusals and local filesystem/decrypt errors are permanent;
|
|
// anything else is unknown. Both the error code and message are inspected.
|
|
const classify = (err: unknown): FailureClass => {
|
|
const e = err as NodeJS.ErrnoException;
|
|
const text =
|
|
`${e?.code ?? ""} ${err instanceof Error ? err.message : String(err)}`.toLowerCase();
|
|
if (
|
|
/timeout|timed out|econnreset|econnrefused|econnaborted|network|socket|eai_again|throttl|temporarily|429|500|502|503|504/.test(
|
|
text,
|
|
)
|
|
) {
|
|
return "transient";
|
|
}
|
|
if (
|
|
/enoent|eacces|eperm|eexist|eisdir|enotempty|erofs|enospc|not found|forbidden|unauthor|decrypt|truncat|401|403|404/.test(
|
|
text,
|
|
)
|
|
) {
|
|
return "permanent";
|
|
}
|
|
return "unknown";
|
|
};
|
|
|
|
const errorMessage = (err: unknown): string =>
|
|
err instanceof Error ? err.message : String(err);
|
|
|
|
// Ensure `linkPath` is a symlink to `target`, rebuilding a missing, wrong, or
|
|
// non-symlink entry. Throws on failure (a directory in the way, no permission)
|
|
// so the caller records it and moves on rather than aborting the run.
|
|
const rebuildSymlink = (linkPath: string, target: string): void => {
|
|
try {
|
|
const st = lstatSync(linkPath);
|
|
if (st.isSymbolicLink() && readlinkSync(linkPath) === target) return;
|
|
} catch {
|
|
// Nothing there (or unreadable): fall through to create it.
|
|
}
|
|
// Remove a wrong symlink or stray file. `force` ignores a missing path but
|
|
// still refuses a directory (no `recursive`), which surfaces as a failure.
|
|
rmSync(linkPath, { force: true });
|
|
symlinkSync(target, linkPath);
|
|
};
|
|
|
|
// The on-disk names for the entries of one directory, in entry order. Each
|
|
// name is used as is unless another entry would get the same name, ignoring
|
|
// case (two names that differ only in case are one entry on a case-insensitive
|
|
// file system); then every entry sharing it gets ` (<id>)`, before the
|
|
// extension when `beforeExtension` is set. A name with an ID added can match
|
|
// another entry's own name (`IMG (6).JPG`), so this repeats until no name is
|
|
// shared. IDs are stable, so the names are too.
|
|
const uniqueNames = (
|
|
entries: { id: number; name: string }[],
|
|
beforeExtension: boolean,
|
|
): string[] => {
|
|
const withID = (id: number, name: string): string => {
|
|
const ext = beforeExtension ? extname(name) : "";
|
|
const stem = name.slice(0, name.length - ext.length);
|
|
return `${stem} (${id})${ext}`;
|
|
};
|
|
const names = entries.map((e) => e.name);
|
|
const suffixed = new Set<number>();
|
|
for (;;) {
|
|
const counts = new Map<string, number>();
|
|
for (const name of names) {
|
|
const key = name.toLowerCase();
|
|
counts.set(key, (counts.get(key) ?? 0) + 1);
|
|
}
|
|
let changed = false;
|
|
for (const [i, { id, name }] of entries.entries()) {
|
|
if (suffixed.has(i)) continue;
|
|
if (counts.get(name.toLowerCase()) === 1) continue;
|
|
names[i] = withID(id, name);
|
|
suffixed.add(i);
|
|
changed = true;
|
|
}
|
|
if (!changed) return names;
|
|
}
|
|
};
|
|
|
|
// The links a file gets in its album's folder: one named after its title, to
|
|
// its original if that is stored. A stored live photo gets two, to its image
|
|
// and its video, each named after the title with that file's extension.
|
|
const linksFor = (
|
|
file: EnteFile,
|
|
stored: { path: string; videoPath?: string } | undefined,
|
|
): { id: number; name: string; file: EnteFile; target?: string }[] => {
|
|
const name = sanitizeFileName(file.metadata.title, `file-${file.id}`);
|
|
if (stored?.videoPath === undefined) {
|
|
return [{ id: file.id, name, file, target: stored?.path }];
|
|
}
|
|
return [stored.path, stored.videoPath].map((target) => ({
|
|
id: file.id,
|
|
name: withExtension(name, extname(target)),
|
|
file,
|
|
target,
|
|
}));
|
|
};
|
|
|
|
// Every date folder (`YYYY/YYYY-MM/YYYY-MM-DD/`) under `root`, whether or not a
|
|
// file in this backup is saved there. A folder that cannot be read is skipped.
|
|
const dateFolders = (root: string): string[] => {
|
|
const subfolders = (dir: string, name: RegExp): string[] => {
|
|
try {
|
|
return readdirSync(dir, { withFileTypes: true })
|
|
.filter((e) => e.isDirectory() && name.test(e.name))
|
|
.map((e) => join(dir, e.name));
|
|
} catch {
|
|
return [];
|
|
}
|
|
};
|
|
return subfolders(root, /^\d{4}$/)
|
|
.flatMap((year) => subfolders(year, /^\d{4}-\d\d$/))
|
|
.flatMap((month) => subfolders(month, /^\d{4}-\d\d-\d\d$/));
|
|
};
|
|
|
|
// Whether the entry at `path` is a symlink a backup to `root` made: one to an
|
|
// original in a `YYYY/YYYY-MM/YYYY-MM-DD/` folder of `root`.
|
|
const linksToOriginal = (path: string, root: string): boolean => {
|
|
if (!lstatSync(path).isSymbolicLink()) return false;
|
|
const target = relative(root, resolve(dirname(path), readlinkSync(path)));
|
|
return /^\d{4}\/\d{4}-\d\d\/\d{4}-\d\d-\d\d\/[^/]+$/.test(target);
|
|
};
|
|
|
|
// Remove the symlinks in the album directory `dir` that point to an original
|
|
// in `root` and are not named in `keep`. Nothing else in the directory is
|
|
// touched: anything else there was put there by the user.
|
|
const removeStaleLinks = (
|
|
dir: string,
|
|
keep: Set<string>,
|
|
root: string,
|
|
): void => {
|
|
for (const name of readdirSync(dir)) {
|
|
if (keep.has(name)) continue;
|
|
const path = join(dir, name);
|
|
if (linksToOriginal(path, root)) rmSync(path);
|
|
}
|
|
};
|
|
|
|
// Remove the directories under `collectionsDir` that an earlier run wrote for
|
|
// an album that is gone or renamed: a directory not named in `current` with a
|
|
// `<name>.json` beside it holding an album ID, which is what a run writes. Its
|
|
// symlinks to originals in `root` are removed; if that leaves it empty, it and
|
|
// its JSON are deleted, otherwise both stay for what the user put there.
|
|
const removeStaleAlbumDirs = (
|
|
collectionsDir: string,
|
|
current: Set<string>,
|
|
root: string,
|
|
): void => {
|
|
for (const entry of readdirSync(collectionsDir, { withFileTypes: true })) {
|
|
if (!entry.isDirectory() || current.has(entry.name)) continue;
|
|
const jsonPath = join(collectionsDir, `${entry.name}.json`);
|
|
try {
|
|
const album = JSON.parse(readFileSync(jsonPath, "utf-8")) as {
|
|
id?: unknown;
|
|
};
|
|
if (typeof album.id !== "number") continue;
|
|
} catch {
|
|
continue;
|
|
}
|
|
const dir = join(collectionsDir, entry.name);
|
|
removeStaleLinks(dir, new Set(), root);
|
|
if (readdirSync(dir).length > 0) continue;
|
|
rmdirSync(dir);
|
|
rmSync(jsonPath);
|
|
}
|
|
};
|
|
|
|
const loadLedger = (path: string): Map<number, FailureEntry> => {
|
|
const ledger = new Map<number, FailureEntry>();
|
|
try {
|
|
const parsed = JSON.parse(readFileSync(path, "utf-8")) as {
|
|
files?: Record<string, FailureEntry>;
|
|
};
|
|
for (const entry of Object.values(parsed.files ?? {})) {
|
|
if (entry && typeof entry.fileID === "number") {
|
|
ledger.set(entry.fileID, entry);
|
|
}
|
|
}
|
|
} catch {
|
|
// No ledger yet, or an unreadable one: start clean.
|
|
}
|
|
return ledger;
|
|
};
|
|
|
|
const saveLedger = (path: string, ledger: Map<number, FailureEntry>): void => {
|
|
if (ledger.size === 0) {
|
|
rmSync(path, { force: true });
|
|
return;
|
|
}
|
|
const files: Record<string, FailureEntry> = {};
|
|
for (const [fileID, entry] of ledger) files[String(fileID)] = entry;
|
|
writeFileSync(
|
|
path,
|
|
JSON.stringify({ version: LEDGER_VERSION, files }, null, 2),
|
|
);
|
|
};
|
|
|
|
// A file's EXIF, XMP and dimensions as its JSON holds them: what
|
|
// `extractImageMetadata` found in its original, or why the original could not
|
|
// be read.
|
|
interface ImageMetadata {
|
|
imageMetadata?: Record<string, unknown>;
|
|
imageMetadataError?: string;
|
|
}
|
|
|
|
// The image metadata for the file whose original is at `originalPath` (for a
|
|
// live photo, its image) and whose JSON is at `jsonPath`. A video gets none,
|
|
// as `photo.exif()` reads none. An original stored before this run is not read
|
|
// again when its JSON already holds image metadata: that is kept. A failed
|
|
// read gives the reason, and fails neither the file nor the run. An original
|
|
// with no EXIF, XMP or JPEG dimensions gets `{}`, so it is not read again.
|
|
const imageMetadataFor = async (
|
|
file: EnteFile,
|
|
originalPath: string,
|
|
jsonPath: string,
|
|
storedThisRun: boolean,
|
|
): Promise<ImageMetadata> => {
|
|
if (file.metadata.fileType === "video") return {};
|
|
if (!storedThisRun) {
|
|
try {
|
|
const { imageMetadata, imageMetadataError } = JSON.parse(
|
|
readFileSync(jsonPath, "utf-8"),
|
|
) as ImageMetadata;
|
|
if (imageMetadata !== undefined || imageMetadataError !== undefined)
|
|
return { imageMetadata, imageMetadataError };
|
|
} catch {
|
|
// No JSON yet, or one that cannot be parsed: read the original.
|
|
}
|
|
}
|
|
try {
|
|
const bytes = await readFile(originalPath);
|
|
return { imageMetadata: extractImageMetadata(bytes) ?? {} };
|
|
} catch (err) {
|
|
return { imageMetadataError: errorMessage(err) };
|
|
}
|
|
};
|
|
|
|
// The file's JSON: its basic fields, its magic metadata, its ML data or the
|
|
// reason the ML data is missing, and its image metadata.
|
|
const writeSidecar = (
|
|
path: string,
|
|
file: EnteFile,
|
|
ml: { mlData?: MLData; mlDataError?: string },
|
|
image: ImageMetadata,
|
|
): void => {
|
|
const meta: Record<string, unknown> = {
|
|
id: file.id,
|
|
collectionID: file.collectionID,
|
|
ownerID: file.ownerID,
|
|
metadata: file.metadata,
|
|
updationTime: file.updationTime,
|
|
};
|
|
if (file.magicMetadata) meta.magicMetadata = file.magicMetadata;
|
|
if (file.pubMagicMetadata) meta.pubMagicMetadata = file.pubMagicMetadata;
|
|
if (ml.mlData) meta.mlData = ml.mlData;
|
|
if (ml.mlDataError) meta.mlDataError = ml.mlDataError;
|
|
if (image.imageMetadata) meta.imageMetadata = image.imageMetadata;
|
|
if (image.imageMetadataError) {
|
|
meta.imageMetadataError = image.imageMetadataError;
|
|
}
|
|
writeFileSync(path, JSON.stringify(meta, null, 2));
|
|
};
|
|
|
|
// The album's JSON: its basic fields, its magic metadata, and its files.
|
|
const writeAlbumJSON = (
|
|
path: string,
|
|
c: Collection,
|
|
files: { id: number; metadata: FileMetadata }[],
|
|
): void => {
|
|
const album: Record<string, unknown> = {
|
|
id: c.id,
|
|
name: c.name,
|
|
type: c.type,
|
|
ownerID: c.ownerID,
|
|
isShared: c.isShared,
|
|
updationTime: c.updationTime,
|
|
};
|
|
if (c.magicMetadata) album.magicMetadata = c.magicMetadata;
|
|
if (c.pubMagicMetadata) album.pubMagicMetadata = c.pubMagicMetadata;
|
|
if (c.sharedMagicMetadata) {
|
|
album.sharedMagicMetadata = c.sharedMagicMetadata;
|
|
}
|
|
album.files = files;
|
|
writeFileSync(path, JSON.stringify(album, null, 2));
|
|
};
|
|
|
|
// The backup itself, which `runBackup` below runs while it holds the lock.
|
|
const runLockedBackup = async (
|
|
lib: BackupLibrary,
|
|
opts: BackupOptions,
|
|
downloadDirectory: string,
|
|
): Promise<BackupResult> => {
|
|
const includeOriginals = opts.includeOriginals ?? true;
|
|
const includeThumbnails = opts.includeThumbnails ?? false;
|
|
const log = opts.onProgress ?? (() => {});
|
|
const only = opts.onlyAlbumNames ? new Set(opts.onlyAlbumNames) : undefined;
|
|
|
|
log("Refreshing library...");
|
|
await lib.refresh();
|
|
|
|
const collectionsDir = join(downloadDirectory, "collections");
|
|
const thumbnailsDir = join(downloadDirectory, "thumbnails");
|
|
mkdirSync(collectionsDir, { recursive: true });
|
|
const { email, userID } = lib.whoami();
|
|
writeFileSync(
|
|
join(downloadDirectory, "account.json"),
|
|
JSON.stringify({ email, userID }, null, 2),
|
|
);
|
|
if (includeThumbnails) mkdirSync(thumbnailsDir, { recursive: true });
|
|
removeLeftoverTempFiles(thumbnailsDir);
|
|
for (const dir of dateFolders(downloadDirectory)) {
|
|
removeLeftoverTempFiles(dir);
|
|
}
|
|
|
|
const ledgerPath = join(downloadDirectory, "failures.json");
|
|
const ledger = loadLedger(ledgerPath);
|
|
const now = Date.now();
|
|
|
|
// Collections in scope, and the distinct files across them (a file shared
|
|
// by two albums is one original). Each file is the membership
|
|
// `representative` picks from all of its albums, in scope or not, so it is
|
|
// saved at the path `photo.savePath` names.
|
|
const allCollections = lib.listCollections();
|
|
const collections = allCollections.filter((c) =>
|
|
only ? only.has(c.name) : true,
|
|
);
|
|
const collectionName = new Map<number, string>();
|
|
for (const c of allCollections) collectionName.set(c.id, c.name);
|
|
|
|
const memberships = new Map<number, EnteFile[]>();
|
|
const filesByCollection = new Map<number, EnteFile[]>();
|
|
for (const c of allCollections) {
|
|
const files = lib.listFiles(c.id);
|
|
filesByCollection.set(c.id, files);
|
|
for (const f of files) {
|
|
const arr = memberships.get(f.id);
|
|
if (arr) arr.push(f);
|
|
else memberships.set(f.id, [f]);
|
|
}
|
|
}
|
|
const distinct = new Map<number, EnteFile>();
|
|
for (const c of collections) {
|
|
for (const f of filesByCollection.get(c.id)!) {
|
|
if (!distinct.has(f.id)) {
|
|
distinct.set(f.id, representative(memberships.get(f.id)!));
|
|
}
|
|
}
|
|
}
|
|
|
|
const errors: BackupError[] = [];
|
|
const failedThisRun = new Set<number>();
|
|
const storedThisRun = new Set<number>();
|
|
let downloaded = 0;
|
|
let skipped = 0;
|
|
|
|
const recordFailure = (
|
|
file: EnteFile,
|
|
collection: string,
|
|
err: unknown,
|
|
): void => {
|
|
// Count at most one attempt per file per run: a file whose original
|
|
// and thumbnail both fail this run must not double its attempt count
|
|
// or appear twice in errors.
|
|
if (failedThisRun.has(file.id)) return;
|
|
const error = errorMessage(err);
|
|
errors.push({
|
|
fileID: file.id,
|
|
title: file.metadata.title,
|
|
collection,
|
|
error,
|
|
});
|
|
const prior = ledger.get(file.id);
|
|
ledger.set(file.id, {
|
|
fileID: file.id,
|
|
title: file.metadata.title,
|
|
classification: classify(err),
|
|
attempts: (prior?.attempts ?? 0) + 1,
|
|
lastTriedAt: now,
|
|
error,
|
|
});
|
|
failedThisRun.add(file.id);
|
|
};
|
|
|
|
// Phase 1: get the bytes. Put each pending original at its save path
|
|
// through the content cache/pools, as `Photo.download()` does, and fetch
|
|
// the optional thumbnails; a present file is left as is.
|
|
if (includeOriginals) {
|
|
for (const [fileID, file] of distinct) {
|
|
if (storedAtSavePath(downloadDirectory, file) !== undefined) {
|
|
skipped++;
|
|
continue;
|
|
}
|
|
try {
|
|
log(`Fetching original ${file.metadata.title} (${fileID})...`);
|
|
// A fetched original is written straight to its save path (a
|
|
// live photo beside it); only one that was already cached
|
|
// elsewhere is copied.
|
|
await placeOriginal(downloadDirectory, file, (dest) =>
|
|
lib.original(fileID, dest),
|
|
);
|
|
storedThisRun.add(fileID);
|
|
downloaded++;
|
|
} catch (err) {
|
|
log(
|
|
`FAILED original ${file.metadata.title}: ${errorMessage(err)}`,
|
|
);
|
|
recordFailure(
|
|
file,
|
|
collectionName.get(file.collectionID) ?? "",
|
|
err,
|
|
);
|
|
}
|
|
}
|
|
}
|
|
|
|
if (includeThumbnails) {
|
|
for (const [fileID, file] of distinct) {
|
|
const dest = join(thumbnailsDir, `${fileID}.jpg`);
|
|
if (isPresent(dest)) continue;
|
|
try {
|
|
const { path } = await lib.thumbnail(fileID);
|
|
await copyAtomic(path, dest);
|
|
} catch (err) {
|
|
recordFailure(
|
|
file,
|
|
collectionName.get(file.collectionID) ?? "",
|
|
err,
|
|
);
|
|
}
|
|
}
|
|
}
|
|
|
|
// Phase 2: rebuild the derived views from the model. Sidecars first, for
|
|
// every present original (this repairs stale ones), each with the file's
|
|
// ML data once an ML data fetch has completed, and its image metadata.
|
|
// When the fetch fails, a file with no cached ML data gets the reason
|
|
// instead and is recorded as failed. The next run fetches its ML data
|
|
// again because none is cached.
|
|
if (includeOriginals) {
|
|
let mlDataError: string | undefined;
|
|
try {
|
|
log("Fetching ML data...");
|
|
await lib.fetchMLData();
|
|
} catch (err) {
|
|
mlDataError = errorMessage(err);
|
|
log(`FAILED ML data: ${mlDataError}`);
|
|
}
|
|
for (const file of distinct.values()) {
|
|
const stored = storedAtSavePath(downloadDirectory, file);
|
|
if (stored === undefined) continue;
|
|
const path = withExtension(
|
|
savePath(downloadDirectory, file),
|
|
".json",
|
|
);
|
|
const image = await imageMetadataFor(
|
|
file,
|
|
stored.path,
|
|
path,
|
|
storedThisRun.has(file.id),
|
|
);
|
|
const mlData = await lib.mlData(file.id);
|
|
if (mlData === undefined && mlDataError !== undefined) {
|
|
writeSidecar(path, file, { mlDataError }, image);
|
|
recordFailure(
|
|
file,
|
|
collectionName.get(file.collectionID) ?? "",
|
|
new Error(`ML data: ${mlDataError}`),
|
|
);
|
|
} else {
|
|
writeSidecar(path, file, { mlData }, image);
|
|
}
|
|
}
|
|
}
|
|
|
|
// Then the per-collection symlink trees and JSON. Directory names are
|
|
// chosen across every album, not just those in scope, so a scoped run
|
|
// names an album the same as a full one and never takes the directory of
|
|
// an album it skipped. Stale entries are removed before anything is
|
|
// rebuilt, so on a case-insensitive file system removing an old name can
|
|
// never remove the new one.
|
|
const dirNames = uniqueNames(
|
|
allCollections.map((c) => ({
|
|
id: c.id,
|
|
name: sanitizeFileName(c.name, `collection-${c.id}`),
|
|
})),
|
|
false,
|
|
);
|
|
const albumDirNames = new Map(
|
|
allCollections.map((c, i) => [c.id, dirNames[i]!]),
|
|
);
|
|
try {
|
|
removeStaleAlbumDirs(
|
|
collectionsDir,
|
|
new Set(dirNames),
|
|
downloadDirectory,
|
|
);
|
|
} catch (err) {
|
|
log(`FAILED removing old album directories: ${errorMessage(err)}`);
|
|
}
|
|
|
|
for (const c of collections) {
|
|
const colDirName = albumDirNames.get(c.id)!;
|
|
const colDir = join(collectionsDir, colDirName);
|
|
mkdirSync(colDir, { recursive: true });
|
|
|
|
// Every album links the one original, saved from the file's entry in
|
|
// `distinct`.
|
|
const files = filesByCollection.get(c.id) ?? [];
|
|
const links = files.flatMap((f) =>
|
|
linksFor(
|
|
f,
|
|
storedAtSavePath(downloadDirectory, distinct.get(f.id)!),
|
|
),
|
|
);
|
|
const linkNames = uniqueNames(links, true);
|
|
try {
|
|
removeStaleLinks(colDir, new Set(linkNames), downloadDirectory);
|
|
} catch (err) {
|
|
log(`FAILED removing old links in ${c.name}: ${errorMessage(err)}`);
|
|
}
|
|
|
|
const metaFiles = files.map((f) => ({
|
|
id: f.id,
|
|
metadata: f.metadata,
|
|
}));
|
|
for (const [i, link] of links.entries()) {
|
|
if (!includeOriginals || link.target === undefined) continue;
|
|
const linkName = linkNames[i]!;
|
|
try {
|
|
rebuildSymlink(
|
|
join(colDir, linkName),
|
|
relative(colDir, link.target),
|
|
);
|
|
} catch (err) {
|
|
log(
|
|
`FAILED symlink ${c.name}/${linkName}: ${errorMessage(err)}`,
|
|
);
|
|
recordFailure(link.file, c.name, err);
|
|
}
|
|
}
|
|
|
|
writeAlbumJSON(
|
|
join(collectionsDir, `${colDirName}.json`),
|
|
c,
|
|
metaFiles,
|
|
);
|
|
}
|
|
|
|
// Reconcile the ledger against what this run actually attempted: an entry
|
|
// survives only for a file that failed this run. A file that succeeded had
|
|
// its failure resolved; a file gone from the library (deleted) or outside
|
|
// this run's scope is not something this run can resolve, so keeping its
|
|
// stale entry would keep the exit code non-zero forever — a single
|
|
// since-deleted photo would fail every future scheduled backup.
|
|
for (const fileID of [...ledger.keys()]) {
|
|
if (!failedThisRun.has(fileID)) ledger.delete(fileID);
|
|
}
|
|
saveLedger(ledgerPath, ledger);
|
|
|
|
return {
|
|
totalFiles: distinct.size,
|
|
downloaded,
|
|
skipped,
|
|
failed: ledger.size,
|
|
errors,
|
|
};
|
|
};
|
|
|
|
// Only one backup of a directory runs at a time, in this process or another:
|
|
// a second one fails at once with an error whose `code` is `ELOCKED`. The lock
|
|
// is the directory `backup.lock`, whose modification time proper-lockfile
|
|
// keeps current while the backup runs. One it has not touched for 10 seconds
|
|
// was left by a run that was killed, and is taken over.
|
|
export const runBackup = async (
|
|
lib: BackupLibrary,
|
|
opts: BackupOptions,
|
|
): Promise<BackupResult> => {
|
|
const downloadDirectory = opts.downloadDirectory;
|
|
if (!downloadDirectory) {
|
|
throw new Error(
|
|
"backup requires a downloadDirectory (pass one to backup() or " +
|
|
"open the library with one)",
|
|
);
|
|
}
|
|
mkdirSync(downloadDirectory, { recursive: true });
|
|
let release: () => Promise<void>;
|
|
try {
|
|
release = await lockfile.lock(downloadDirectory, {
|
|
lockfilePath: join(downloadDirectory, "backup.lock"),
|
|
});
|
|
} catch (err) {
|
|
if ((err as NodeJS.ErrnoException).code !== "ELOCKED") throw err;
|
|
throw Object.assign(
|
|
new Error(`another backup of ${downloadDirectory} is running`),
|
|
{ code: "ELOCKED" },
|
|
);
|
|
}
|
|
try {
|
|
return await runLockedBackup(lib, opts, downloadDirectory);
|
|
} finally {
|
|
await release();
|
|
}
|
|
};
|