quak backup writes each original's EXIF, XMP and dimensions into its JSON (closes #167)
check / check (push) Successful in 3m18s

Each file's JSON gains imageMetadata, what extractImageMetadata finds in
the stored original (for a live photo, its image), or the reason the
read failed in imageMetadataError; a failed read fails neither the file
nor the run. A video is not read. An original is read when the run
stores it or when its JSON has neither field; otherwise the field is
carried over from that JSON, so a run does not read every original
again. The hand-built JPEG fixtures move to test/exif-jpeg.ts so the
backup tests can use them.

Judgement call: an original with nothing to record gets imageMetadata
{} instead of no field, so it is not read again on every run.

Model: opus-5-5
This commit was merged in pull request #177.
This commit is contained in:
2026-10-06 16:47:31 +02:00
parent f6317109bc
commit 31b50a211d
7 changed files with 371 additions and 106 deletions
+72 -12
View File
@@ -11,7 +11,9 @@
// YYYY/YYYY-MM/YYYY-MM-DD/
// YYYY-MM-DD.<fileID>.<ext> the decrypted bytes (the save path)
// YYYY-MM-DD.<fileID>.json per-file metadata sidecar, with
// the file's ML data
// the file's ML data and its
// original's EXIF, XMP and
// dimensions
// collections/<name>/<title> symlink to the original
// collections/<name>.json per-collection metadata
// account.json the account's email and user ID
@@ -28,7 +30,10 @@
// no unique state, so they are rebuilt every run; that repairs stale sidecars
// and missing or broken symlinks left by an earlier crash. A rebuild also
// removes the symlinks to originals that no longer belong to an album, and the
// directories of albums that no longer exist.
// directories of albums that no longer exist. The one thing a sidecar takes
// from the sidecar it replaces is its original's EXIF, XMP and dimensions (or
// why they could not be read), so that a run does not read every stored
// original again; a sidecar without them gets them read from the original.
//
// Resilience (issue #8): no per-file condition aborts the run. A failed
// download, a failed symlink, or ML data missing because the ML data fetch
@@ -53,6 +58,7 @@ import {
symlinkSync,
writeFileSync,
} from "node:fs";
import { readFile } from "node:fs/promises";
import { dirname, extname, join, relative, resolve } from "node:path";
import { removeLeftoverTempFiles } from "./download/index.js";
@@ -64,6 +70,7 @@ import {
storedAtSavePath,
} from "./library/content.js";
import { representative } from "./library/records.js";
import { extractImageMetadata } from "./metadata-backup.js";
import type { MLData } from "./mldata-fetch.js";
import type { Collection, EnteFile, FileMetadata } from "./model/types.js";
@@ -354,12 +361,53 @@ const saveLedger = (path: string, ledger: Map<number, FailureEntry>): void => {
);
};
// The file's JSON: its basic fields, its magic metadata, and its ML data, or
// the reason the ML data is missing.
// A file's EXIF, XMP and dimensions as its JSON holds them: what
// `extractImageMetadata` found in its original, or why the original could not
// be read.
interface ImageMetadata {
imageMetadata?: Record<string, unknown>;
imageMetadataError?: string;
}
// The image metadata for the file whose original is at `originalPath` (for a
// live photo, its image) and whose JSON is at `jsonPath`. A video gets none,
// as `photo.exif()` reads none. An original stored before this run is not read
// again when its JSON already holds image metadata: that is kept. A failed
// read gives the reason, and fails neither the file nor the run. An original
// with no EXIF, XMP or JPEG dimensions gets `{}`, so it is not read again.
const imageMetadataFor = async (
file: EnteFile,
originalPath: string,
jsonPath: string,
storedThisRun: boolean,
): Promise<ImageMetadata> => {
if (file.metadata.fileType === "video") return {};
if (!storedThisRun) {
try {
const { imageMetadata, imageMetadataError } = JSON.parse(
readFileSync(jsonPath, "utf-8"),
) as ImageMetadata;
if (imageMetadata !== undefined || imageMetadataError !== undefined)
return { imageMetadata, imageMetadataError };
} catch {
// No JSON yet, or one that cannot be parsed: read the original.
}
}
try {
const bytes = await readFile(originalPath);
return { imageMetadata: extractImageMetadata(bytes) ?? {} };
} catch (err) {
return { imageMetadataError: errorMessage(err) };
}
};
// The file's JSON: its basic fields, its magic metadata, its ML data or the
// reason the ML data is missing, and its image metadata.
const writeSidecar = (
path: string,
file: EnteFile,
ml: { mlData?: MLData; mlDataError?: string },
image: ImageMetadata,
): void => {
const meta: Record<string, unknown> = {
id: file.id,
@@ -372,6 +420,10 @@ const writeSidecar = (
if (file.pubMagicMetadata) meta.pubMagicMetadata = file.pubMagicMetadata;
if (ml.mlData) meta.mlData = ml.mlData;
if (ml.mlDataError) meta.mlDataError = ml.mlDataError;
if (image.imageMetadata) meta.imageMetadata = image.imageMetadata;
if (image.imageMetadataError) {
meta.imageMetadataError = image.imageMetadataError;
}
writeFileSync(path, JSON.stringify(meta, null, 2));
};
@@ -468,6 +520,7 @@ export const runBackup = async (
const errors: BackupError[] = [];
const failedThisRun = new Set<number>();
const storedThisRun = new Set<number>();
let downloaded = 0;
let skipped = 0;
@@ -516,6 +569,7 @@ export const runBackup = async (
await placeOriginal(downloadDirectory, file, (dest) =>
lib.original(fileID, dest),
);
storedThisRun.add(fileID);
downloaded++;
} catch (err) {
log(
@@ -549,9 +603,10 @@ export const runBackup = async (
// Phase 2: rebuild the derived views from the model. Sidecars first, for
// every present original (this repairs stale ones), each with the file's
// ML data once an ML data fetch has completed. When the fetch fails, a
// file with no cached ML data gets the reason instead and is recorded as
// failed. The next run fetches its ML data again because none is cached.
// ML data once an ML data fetch has completed, and its image metadata.
// When the fetch fails, a file with no cached ML data gets the reason
// instead and is recorded as failed. The next run fetches its ML data
// again because none is cached.
if (includeOriginals) {
let mlDataError: string | undefined;
try {
@@ -562,23 +617,28 @@ export const runBackup = async (
log(`FAILED ML data: ${mlDataError}`);
}
for (const file of distinct.values()) {
if (storedAtSavePath(downloadDirectory, file) === undefined) {
continue;
}
const stored = storedAtSavePath(downloadDirectory, file);
if (stored === undefined) continue;
const path = withExtension(
savePath(downloadDirectory, file),
".json",
);
const image = await imageMetadataFor(
file,
stored.path,
path,
storedThisRun.has(file.id),
);
const mlData = await lib.mlData(file.id);
if (mlData === undefined && mlDataError !== undefined) {
writeSidecar(path, file, { mlDataError });
writeSidecar(path, file, { mlDataError }, image);
recordFailure(
file,
collectionName.get(file.collectionID) ?? "",
new Error(`ML data: ${mlDataError}`),
);
} else {
writeSidecar(path, file, { mlData });
writeSidecar(path, file, { mlData }, image);
}
}
}