Files
quak/src/backup.ts
T
clawbot 73846b32fc
check / check (push) Waiting to run
quak backup --verify re-hashes stored originals and downloads again any that do not match (closes #168)
`--verify`, or `lib.backup({ verify: true })`, hashes each original already
at its save path as the download check does, streamed, a live photo as
`<imageHash>:<videoHash>`. A mismatch is logged, removed and fetched again in
the same run; a failed fetch goes into `failures.json`. A file with no
recorded hash counts as unchecked. The result, the summary and `--json` gain
`verified`, `mismatched` and `unchecked`.

Judgement call: a stored original that cannot be read for hashing is recorded as failed and left in place.
Judgement call: the summary prints the three counts only with `--verify`.
Judgement call: the README's `BackupOptions` list does not name `verify`, to stay clear of #178's edit of that paragraph; Backup layout documents it.

Model: opus-5-5
2026-10-06 14:57:36 +00:00

822 lines
31 KiB
TypeScript

// The backup command, rebuilt on the library API (issue #51).
//
// `lib.backup()` waits for a completed refresh of the library (a failed one
// fails the backup before any file is touched), then, for every file in scope,
// puts its original at its save path under `downloadDirectory`, as
// `Photo.download()` does, waits for an ML data fetch, and rebuilds the derived
// views (per-file sidecars, per-collection symlink trees, per-collection JSON)
// from the model. The on-disk layout:
//
// <downloadDirectory>/
// YYYY/YYYY-MM/YYYY-MM-DD/
// YYYY-MM-DD.<fileID>.<ext> the decrypted bytes (the save path)
// YYYY-MM-DD.<fileID>.json per-file metadata sidecar, with
// the file's ML data and its
// original's EXIF, XMP and
// dimensions
// collections/<name>/<title> symlink to the original
// collections/<name>.json per-collection metadata
// account.json the account's email and user ID
// failures.json durable ledger of unresolved failures
//
// A live photo's original is its image and its video, each with its own
// extension, beside `YYYY-MM-DD.<fileID>.livephoto.json` naming them; its album
// folders link both.
//
// Crash-safety rests on two properties. Bytes are present-means-complete: an
// original appears at its save path only via the content layer's atomic
// temp-then-rename, so a file that exists is whole and is never re-fetched — an
// interrupted run resumes by looking at the save paths. The derived views hold
// no unique state, so they are rebuilt every run; that repairs stale sidecars
// and missing or broken symlinks left by an earlier crash. A rebuild also
// removes the symlinks to originals that no longer belong to an album, and the
// directories of albums that no longer exist. The one thing a sidecar takes
// from the sidecar it replaces is its original's EXIF, XMP and dimensions (or
// why they could not be read), so that a run does not read every stored
// original again; a sidecar without them gets them read from the original.
//
// With `verify`, each original already at its save path is hashed as the
// download check hashes it, and one that does not match the content hash its
// metadata records is removed and fetched again in the same run.
//
// Resilience (issue #8): no per-file condition aborts the run. A failed
// download, a failed symlink, or ML data missing because the ML data fetch
// failed is caught, recorded in `failures.json` with a classification, a
// running attempt count, and the last-tried time, and the run continues.
// `result.failed` — and thus the CLI's exit code — stays non-zero while any
// failure remains unresolved and clears once every one succeeds. Each run
// reconciles the ledger against the files it attempted, so an entry for a file
// that has since left the library (deleted) or this run's scope is dropped
// rather than counted forever, which would poison a scheduled backup's exit
// code.
import {
createReadStream,
lstatSync,
mkdirSync,
readdirSync,
readFileSync,
readlinkSync,
rmdirSync,
rmSync,
statSync,
symlinkSync,
writeFileSync,
} from "node:fs";
import { readFile } from "node:fs/promises";
import { dirname, extname, join, relative, resolve } from "node:path";
import {
chunkHashFinal,
chunkHashInit,
chunkHashUpdate,
init,
} from "./crypto/index.js";
import { removeLeftoverTempFiles } from "./download/index.js";
import { sanitizeFileName, withExtension } from "./filename.js";
import {
copyAtomic,
placeOriginal,
savePath,
storedAtSavePath,
} from "./library/content.js";
import { representative } from "./library/records.js";
import { extractImageMetadata } from "./metadata-backup.js";
import type { MLData } from "./mldata-fetch.js";
import type { Collection, EnteFile, FileMetadata } from "./model/types.js";
export type ProgressCallback = (message: string) => void;
export interface BackupOptions {
// Where the backup tree lives. `lib.backup()` defaults it to the library's
// download directory; `runBackup` with none throws before any network
// traffic.
downloadDirectory?: string;
// Fetch and store full-resolution originals. Default true.
includeOriginals?: boolean;
// Also fetch and store thumbnails under `thumbnails/<fileID>.jpg`. Default
// false.
includeThumbnails?: boolean;
// Restrict the backup to albums with these names; others are left untouched.
onlyAlbumNames?: string[];
// Hash each original already at its save path, and fetch again any whose
// bytes do not match the content hash its metadata records. Default false.
verify?: boolean;
onProgress?: ProgressCallback;
}
export interface BackupError {
fileID: number;
title: string;
collection: string;
error: string;
}
export interface BackupResult {
// Distinct files in scope this run.
totalFiles: number;
// Originals fetched (or copied from the cache) this run.
downloaded: number;
// Originals already at their save path and left untouched.
skipped: number;
// With `verify`, the originals already at their save path whose hash
// matched, those whose hash did not (each removed and fetched again), and
// those whose metadata records no hash (left as they are). All three are
// zero without `verify`.
verified: number;
mismatched: number;
unchecked: number;
// Files with an unresolved failure after this run (the ledger size); the
// CLI exits non-zero while this is above zero. A file can be both
// downloaded and failed if its bytes landed but its symlink did not.
failed: number;
// This run's per-file errors, in encounter order.
errors: BackupError[];
}
// The slice of the library that backup drives. `Library` implements it; a test
// can drive backup with a stand-in.
export interface BackupLibrary {
// The account the library belongs to.
whoami(): { email: string; userID: number };
refresh(): Promise<void>;
listCollections(): Collection[];
listFiles(collectionID: number): EnteFile[];
// Get an original's bytes onto disk through the content cache/pools,
// returning where they landed: `destination` when they were fetched now,
// otherwise wherever they already were (the cache, or the library's save
// path). A live photo lands as its image and its video, fetched now beside
// `destination`.
original(
fileID: number,
destination: string,
): Promise<{ path: string; videoPath?: string }>;
thumbnail(fileID: number): Promise<{ path: string }>;
// Wait for an ML data fetch to complete, joining one already running or
// starting one. Rejects with the reason when it fails; resolves at once
// when the library cannot fetch ML data.
fetchMLData(): Promise<void>;
// A file's cached ML data, as `lib.mldata.forFile()` returns it, or
// undefined when none is cached.
mlData(fileID: number): Promise<MLData | undefined>;
}
type FailureClass = "transient" | "permanent" | "unknown";
interface FailureEntry {
fileID: number;
title: string;
classification: FailureClass;
attempts: number;
lastTriedAt: number;
error: string;
}
const LEDGER_VERSION = 1;
// A regular file with content is treated as complete. A zero-byte file is not:
// it is the shape an aborted write leaves and must be re-fetched.
const isPresent = (path: string): boolean => {
try {
const s = statSync(path);
return s.isFile() && s.size > 0;
} catch {
return false;
}
};
// Best-effort classification for the ledger. Retryable server/network problems
// are transient; refusals and local filesystem/decrypt errors are permanent;
// anything else is unknown. Both the error code and message are inspected.
const classify = (err: unknown): FailureClass => {
const e = err as NodeJS.ErrnoException;
const text =
`${e?.code ?? ""} ${err instanceof Error ? err.message : String(err)}`.toLowerCase();
if (
/timeout|timed out|econnreset|econnrefused|econnaborted|network|socket|eai_again|throttl|temporarily|429|500|502|503|504/.test(
text,
)
) {
return "transient";
}
if (
/enoent|eacces|eperm|eexist|eisdir|enotempty|erofs|enospc|not found|forbidden|unauthor|decrypt|truncat|401|403|404/.test(
text,
)
) {
return "permanent";
}
return "unknown";
};
const errorMessage = (err: unknown): string =>
err instanceof Error ? err.message : String(err);
// Ensure `linkPath` is a symlink to `target`, rebuilding a missing, wrong, or
// non-symlink entry. Throws on failure (a directory in the way, no permission)
// so the caller records it and moves on rather than aborting the run.
const rebuildSymlink = (linkPath: string, target: string): void => {
try {
const st = lstatSync(linkPath);
if (st.isSymbolicLink() && readlinkSync(linkPath) === target) return;
} catch {
// Nothing there (or unreadable): fall through to create it.
}
// Remove a wrong symlink or stray file. `force` ignores a missing path but
// still refuses a directory (no `recursive`), which surfaces as a failure.
rmSync(linkPath, { force: true });
symlinkSync(target, linkPath);
};
// The on-disk names for the entries of one directory, in entry order. Each
// name is used as is unless another entry would get the same name, ignoring
// case (two names that differ only in case are one entry on a case-insensitive
// file system); then every entry sharing it gets ` (<id>)`, before the
// extension when `beforeExtension` is set. A name with an ID added can match
// another entry's own name (`IMG (6).JPG`), so this repeats until no name is
// shared. IDs are stable, so the names are too.
const uniqueNames = (
entries: { id: number; name: string }[],
beforeExtension: boolean,
): string[] => {
const withID = (id: number, name: string): string => {
const ext = beforeExtension ? extname(name) : "";
const stem = name.slice(0, name.length - ext.length);
return `${stem} (${id})${ext}`;
};
const names = entries.map((e) => e.name);
const suffixed = new Set<number>();
for (;;) {
const counts = new Map<string, number>();
for (const name of names) {
const key = name.toLowerCase();
counts.set(key, (counts.get(key) ?? 0) + 1);
}
let changed = false;
for (const [i, { id, name }] of entries.entries()) {
if (suffixed.has(i)) continue;
if (counts.get(name.toLowerCase()) === 1) continue;
names[i] = withID(id, name);
suffixed.add(i);
changed = true;
}
if (!changed) return names;
}
};
// The links a file gets in its album's folder: one named after its title, to
// its original if that is stored. A stored live photo gets two, to its image
// and its video, each named after the title with that file's extension.
const linksFor = (
file: EnteFile,
stored: { path: string; videoPath?: string } | undefined,
): { id: number; name: string; file: EnteFile; target?: string }[] => {
const name = sanitizeFileName(file.metadata.title, `file-${file.id}`);
if (stored?.videoPath === undefined) {
return [{ id: file.id, name, file, target: stored?.path }];
}
return [stored.path, stored.videoPath].map((target) => ({
id: file.id,
name: withExtension(name, extname(target)),
file,
target,
}));
};
// Every date folder (`YYYY/YYYY-MM/YYYY-MM-DD/`) under `root`, whether or not a
// file in this backup is saved there. A folder that cannot be read is skipped.
const dateFolders = (root: string): string[] => {
const subfolders = (dir: string, name: RegExp): string[] => {
try {
return readdirSync(dir, { withFileTypes: true })
.filter((e) => e.isDirectory() && name.test(e.name))
.map((e) => join(dir, e.name));
} catch {
return [];
}
};
return subfolders(root, /^\d{4}$/)
.flatMap((year) => subfolders(year, /^\d{4}-\d\d$/))
.flatMap((month) => subfolders(month, /^\d{4}-\d\d-\d\d$/));
};
// Whether the entry at `path` is a symlink a backup to `root` made: one to an
// original in a `YYYY/YYYY-MM/YYYY-MM-DD/` folder of `root`.
const linksToOriginal = (path: string, root: string): boolean => {
if (!lstatSync(path).isSymbolicLink()) return false;
const target = relative(root, resolve(dirname(path), readlinkSync(path)));
return /^\d{4}\/\d{4}-\d\d\/\d{4}-\d\d-\d\d\/[^/]+$/.test(target);
};
// Remove the symlinks in the album directory `dir` that point to an original
// in `root` and are not named in `keep`. Nothing else in the directory is
// touched: anything else there was put there by the user.
const removeStaleLinks = (
dir: string,
keep: Set<string>,
root: string,
): void => {
for (const name of readdirSync(dir)) {
if (keep.has(name)) continue;
const path = join(dir, name);
if (linksToOriginal(path, root)) rmSync(path);
}
};
// Remove the directories under `collectionsDir` that an earlier run wrote for
// an album that is gone or renamed: a directory not named in `current` with a
// `<name>.json` beside it holding an album ID, which is what a run writes. Its
// symlinks to originals in `root` are removed; if that leaves it empty, it and
// its JSON are deleted, otherwise both stay for what the user put there.
const removeStaleAlbumDirs = (
collectionsDir: string,
current: Set<string>,
root: string,
): void => {
for (const entry of readdirSync(collectionsDir, { withFileTypes: true })) {
if (!entry.isDirectory() || current.has(entry.name)) continue;
const jsonPath = join(collectionsDir, `${entry.name}.json`);
try {
const album = JSON.parse(readFileSync(jsonPath, "utf-8")) as {
id?: unknown;
};
if (typeof album.id !== "number") continue;
} catch {
continue;
}
const dir = join(collectionsDir, entry.name);
removeStaleLinks(dir, new Set(), root);
if (readdirSync(dir).length > 0) continue;
rmdirSync(dir);
rmSync(jsonPath);
}
};
const loadLedger = (path: string): Map<number, FailureEntry> => {
const ledger = new Map<number, FailureEntry>();
try {
const parsed = JSON.parse(readFileSync(path, "utf-8")) as {
files?: Record<string, FailureEntry>;
};
for (const entry of Object.values(parsed.files ?? {})) {
if (entry && typeof entry.fileID === "number") {
ledger.set(entry.fileID, entry);
}
}
} catch {
// No ledger yet, or an unreadable one: start clean.
}
return ledger;
};
const saveLedger = (path: string, ledger: Map<number, FailureEntry>): void => {
if (ledger.size === 0) {
rmSync(path, { force: true });
return;
}
const files: Record<string, FailureEntry> = {};
for (const [fileID, entry] of ledger) files[String(fileID)] = entry;
writeFileSync(
path,
JSON.stringify({ version: LEDGER_VERSION, files }, null, 2),
);
};
// The content hash of the original stored at `stored`, computed as the download
// check computes it: over each file's bytes, read in chunks, and for a live
// photo `<imageHash>:<videoHash>`.
const storedHash = async (stored: {
path: string;
videoPath?: string;
}): Promise<string> => {
await init();
const hashFile = async (path: string): Promise<string> => {
const state = chunkHashInit();
for await (const chunk of createReadStream(path)) {
chunkHashUpdate(state, chunk as Buffer);
}
return chunkHashFinal(state);
};
const hash = await hashFile(stored.path);
if (stored.videoPath === undefined) return hash;
return `${hash}:${await hashFile(stored.videoPath)}`;
};
// A file's EXIF, XMP and dimensions as its JSON holds them: what
// `extractImageMetadata` found in its original, or why the original could not
// be read.
interface ImageMetadata {
imageMetadata?: Record<string, unknown>;
imageMetadataError?: string;
}
// The image metadata for the file whose original is at `originalPath` (for a
// live photo, its image) and whose JSON is at `jsonPath`. A video gets none,
// as `photo.exif()` reads none. An original stored before this run is not read
// again when its JSON already holds image metadata: that is kept. A failed
// read gives the reason, and fails neither the file nor the run. An original
// with no EXIF, XMP or JPEG dimensions gets `{}`, so it is not read again.
const imageMetadataFor = async (
file: EnteFile,
originalPath: string,
jsonPath: string,
storedThisRun: boolean,
): Promise<ImageMetadata> => {
if (file.metadata.fileType === "video") return {};
if (!storedThisRun) {
try {
const { imageMetadata, imageMetadataError } = JSON.parse(
readFileSync(jsonPath, "utf-8"),
) as ImageMetadata;
if (imageMetadata !== undefined || imageMetadataError !== undefined)
return { imageMetadata, imageMetadataError };
} catch {
// No JSON yet, or one that cannot be parsed: read the original.
}
}
try {
const bytes = await readFile(originalPath);
return { imageMetadata: extractImageMetadata(bytes) ?? {} };
} catch (err) {
return { imageMetadataError: errorMessage(err) };
}
};
// The file's JSON: its basic fields, its magic metadata, its ML data or the
// reason the ML data is missing, and its image metadata.
const writeSidecar = (
path: string,
file: EnteFile,
ml: { mlData?: MLData; mlDataError?: string },
image: ImageMetadata,
): void => {
const meta: Record<string, unknown> = {
id: file.id,
collectionID: file.collectionID,
ownerID: file.ownerID,
metadata: file.metadata,
updationTime: file.updationTime,
};
if (file.magicMetadata) meta.magicMetadata = file.magicMetadata;
if (file.pubMagicMetadata) meta.pubMagicMetadata = file.pubMagicMetadata;
if (ml.mlData) meta.mlData = ml.mlData;
if (ml.mlDataError) meta.mlDataError = ml.mlDataError;
if (image.imageMetadata) meta.imageMetadata = image.imageMetadata;
if (image.imageMetadataError) {
meta.imageMetadataError = image.imageMetadataError;
}
writeFileSync(path, JSON.stringify(meta, null, 2));
};
// The album's JSON: its basic fields, its magic metadata, and its files.
const writeAlbumJSON = (
path: string,
c: Collection,
files: { id: number; metadata: FileMetadata }[],
): void => {
const album: Record<string, unknown> = {
id: c.id,
name: c.name,
type: c.type,
ownerID: c.ownerID,
isShared: c.isShared,
updationTime: c.updationTime,
};
if (c.magicMetadata) album.magicMetadata = c.magicMetadata;
if (c.pubMagicMetadata) album.pubMagicMetadata = c.pubMagicMetadata;
if (c.sharedMagicMetadata) {
album.sharedMagicMetadata = c.sharedMagicMetadata;
}
album.files = files;
writeFileSync(path, JSON.stringify(album, null, 2));
};
export const runBackup = async (
lib: BackupLibrary,
opts: BackupOptions,
): Promise<BackupResult> => {
const downloadDirectory = opts.downloadDirectory;
if (!downloadDirectory) {
throw new Error(
"backup requires a downloadDirectory (pass one to backup() or " +
"open the library with one)",
);
}
const includeOriginals = opts.includeOriginals ?? true;
const includeThumbnails = opts.includeThumbnails ?? false;
const log = opts.onProgress ?? (() => {});
const only = opts.onlyAlbumNames ? new Set(opts.onlyAlbumNames) : undefined;
const verify = opts.verify ?? false;
log("Refreshing library...");
await lib.refresh();
const collectionsDir = join(downloadDirectory, "collections");
const thumbnailsDir = join(downloadDirectory, "thumbnails");
mkdirSync(collectionsDir, { recursive: true });
const { email, userID } = lib.whoami();
writeFileSync(
join(downloadDirectory, "account.json"),
JSON.stringify({ email, userID }, null, 2),
);
if (includeThumbnails) mkdirSync(thumbnailsDir, { recursive: true });
removeLeftoverTempFiles(thumbnailsDir);
for (const dir of dateFolders(downloadDirectory)) {
removeLeftoverTempFiles(dir);
}
const ledgerPath = join(downloadDirectory, "failures.json");
const ledger = loadLedger(ledgerPath);
const now = Date.now();
// Collections in scope, and the distinct files across them (a file shared
// by two albums is one original). Each file is the membership
// `representative` picks from all of its albums, in scope or not, so it is
// saved at the path `photo.savePath` names.
const allCollections = lib.listCollections();
const collections = allCollections.filter((c) =>
only ? only.has(c.name) : true,
);
const collectionName = new Map<number, string>();
for (const c of allCollections) collectionName.set(c.id, c.name);
const memberships = new Map<number, EnteFile[]>();
const filesByCollection = new Map<number, EnteFile[]>();
for (const c of allCollections) {
const files = lib.listFiles(c.id);
filesByCollection.set(c.id, files);
for (const f of files) {
const arr = memberships.get(f.id);
if (arr) arr.push(f);
else memberships.set(f.id, [f]);
}
}
const distinct = new Map<number, EnteFile>();
for (const c of collections) {
for (const f of filesByCollection.get(c.id)!) {
if (!distinct.has(f.id)) {
distinct.set(f.id, representative(memberships.get(f.id)!));
}
}
}
const errors: BackupError[] = [];
const failedThisRun = new Set<number>();
const storedThisRun = new Set<number>();
let downloaded = 0;
let skipped = 0;
let verified = 0;
let mismatched = 0;
let unchecked = 0;
const recordFailure = (
file: EnteFile,
collection: string,
err: unknown,
): void => {
// Count at most one attempt per file per run: a file whose original
// and thumbnail both fail this run must not double its attempt count
// or appear twice in errors.
if (failedThisRun.has(file.id)) return;
const error = errorMessage(err);
errors.push({
fileID: file.id,
title: file.metadata.title,
collection,
error,
});
const prior = ledger.get(file.id);
ledger.set(file.id, {
fileID: file.id,
title: file.metadata.title,
classification: classify(err),
attempts: (prior?.attempts ?? 0) + 1,
lastTriedAt: now,
error,
});
failedThisRun.add(file.id);
};
// Phase 1: get the bytes. Put each pending original at its save path
// through the content cache/pools, as `Photo.download()` does, and fetch
// the optional thumbnails; a present file is left as is. With `verify`, a
// present original is hashed first, and one that does not match the hash
// its metadata records is removed and fetched like a missing one. One
// that cannot be read is recorded as failed and left where it is.
if (includeOriginals) {
for (const [fileID, file] of distinct) {
let stored = storedAtSavePath(downloadDirectory, file);
if (stored !== undefined && verify) {
try {
if (file.metadata.hash === undefined) {
unchecked++;
} else if (
(await storedHash(stored)) === file.metadata.hash
) {
verified++;
} else {
log(
`MISMATCH original ${file.metadata.title} (${fileID}): its bytes do not match its content hash`,
);
mismatched++;
rmSync(stored.path);
if (stored.videoPath !== undefined) {
rmSync(stored.videoPath);
}
stored = undefined;
}
} catch (err) {
log(
`FAILED verifying original ${file.metadata.title}: ${errorMessage(err)}`,
);
recordFailure(
file,
collectionName.get(file.collectionID) ?? "",
err,
);
continue;
}
}
if (stored !== undefined) {
skipped++;
continue;
}
try {
log(`Fetching original ${file.metadata.title} (${fileID})...`);
// A fetched original is written straight to its save path (a
// live photo beside it); only one that was already cached
// elsewhere is copied.
await placeOriginal(downloadDirectory, file, (dest) =>
lib.original(fileID, dest),
);
storedThisRun.add(fileID);
downloaded++;
} catch (err) {
log(
`FAILED original ${file.metadata.title}: ${errorMessage(err)}`,
);
recordFailure(
file,
collectionName.get(file.collectionID) ?? "",
err,
);
}
}
}
if (includeThumbnails) {
for (const [fileID, file] of distinct) {
const dest = join(thumbnailsDir, `${fileID}.jpg`);
if (isPresent(dest)) continue;
try {
const { path } = await lib.thumbnail(fileID);
await copyAtomic(path, dest);
} catch (err) {
recordFailure(
file,
collectionName.get(file.collectionID) ?? "",
err,
);
}
}
}
// Phase 2: rebuild the derived views from the model. Sidecars first, for
// every present original (this repairs stale ones), each with the file's
// ML data once an ML data fetch has completed, and its image metadata.
// When the fetch fails, a file with no cached ML data gets the reason
// instead and is recorded as failed. The next run fetches its ML data
// again because none is cached.
if (includeOriginals) {
let mlDataError: string | undefined;
try {
log("Fetching ML data...");
await lib.fetchMLData();
} catch (err) {
mlDataError = errorMessage(err);
log(`FAILED ML data: ${mlDataError}`);
}
for (const file of distinct.values()) {
const stored = storedAtSavePath(downloadDirectory, file);
if (stored === undefined) continue;
const path = withExtension(
savePath(downloadDirectory, file),
".json",
);
const image = await imageMetadataFor(
file,
stored.path,
path,
storedThisRun.has(file.id),
);
const mlData = await lib.mlData(file.id);
if (mlData === undefined && mlDataError !== undefined) {
writeSidecar(path, file, { mlDataError }, image);
recordFailure(
file,
collectionName.get(file.collectionID) ?? "",
new Error(`ML data: ${mlDataError}`),
);
} else {
writeSidecar(path, file, { mlData }, image);
}
}
}
// Then the per-collection symlink trees and JSON. Directory names are
// chosen across every album, not just those in scope, so a scoped run
// names an album the same as a full one and never takes the directory of
// an album it skipped. Stale entries are removed before anything is
// rebuilt, so on a case-insensitive file system removing an old name can
// never remove the new one.
const dirNames = uniqueNames(
allCollections.map((c) => ({
id: c.id,
name: sanitizeFileName(c.name, `collection-${c.id}`),
})),
false,
);
const albumDirNames = new Map(
allCollections.map((c, i) => [c.id, dirNames[i]!]),
);
try {
removeStaleAlbumDirs(
collectionsDir,
new Set(dirNames),
downloadDirectory,
);
} catch (err) {
log(`FAILED removing old album directories: ${errorMessage(err)}`);
}
for (const c of collections) {
const colDirName = albumDirNames.get(c.id)!;
const colDir = join(collectionsDir, colDirName);
mkdirSync(colDir, { recursive: true });
// Every album links the one original, saved from the file's entry in
// `distinct`.
const files = filesByCollection.get(c.id) ?? [];
const links = files.flatMap((f) =>
linksFor(
f,
storedAtSavePath(downloadDirectory, distinct.get(f.id)!),
),
);
const linkNames = uniqueNames(links, true);
try {
removeStaleLinks(colDir, new Set(linkNames), downloadDirectory);
} catch (err) {
log(`FAILED removing old links in ${c.name}: ${errorMessage(err)}`);
}
const metaFiles = files.map((f) => ({
id: f.id,
metadata: f.metadata,
}));
for (const [i, link] of links.entries()) {
if (!includeOriginals || link.target === undefined) continue;
const linkName = linkNames[i]!;
try {
rebuildSymlink(
join(colDir, linkName),
relative(colDir, link.target),
);
} catch (err) {
log(
`FAILED symlink ${c.name}/${linkName}: ${errorMessage(err)}`,
);
recordFailure(link.file, c.name, err);
}
}
writeAlbumJSON(
join(collectionsDir, `${colDirName}.json`),
c,
metaFiles,
);
}
// Reconcile the ledger against what this run actually attempted: an entry
// survives only for a file that failed this run. A file that succeeded had
// its failure resolved; a file gone from the library (deleted) or outside
// this run's scope is not something this run can resolve, so keeping its
// stale entry would keep the exit code non-zero forever — a single
// since-deleted photo would fail every future scheduled backup.
for (const fileID of [...ledger.keys()]) {
if (!failedThisRun.has(fileID)) ledger.delete(fileID);
}
saveLedger(ledgerPath, ledger);
return {
totalFiles: distinct.size,
downloaded,
skipped,
verified,
mismatched,
unchecked,
failed: ledger.size,
errors,
};
};