backup-metadata: keep going when an ML data request fails (closes #101)
check / check (push) Successful in 27s

Each ML data request of up to 200 files is now tried on its own. A request
that still fails after its retries is logged, its files are written with the
reason in `mlDataError`, and the command exits 1 once the dump is complete.
`fetchMLData`, used only here, is removed in favour of a per-batch loop over
`fetchMLDataBatch`.

Model: opus-5-5
This commit was merged in pull request #114.
This commit is contained in:
2026-09-23 05:49:38 +02:00
parent d05b53d560
commit bf3b20df2f
6 changed files with 123 additions and 37 deletions
+2 -2
View File
@@ -323,11 +323,11 @@ export const backupMetadataCommand = async (
if (!client) return 1;
const lib = await openReadLibrary(ctx, client);
try {
await runMetadataBackup(lib, client, dir, {
const { failedMLBatches } = await runMetadataBackup(lib, client, dir, {
exif: opts.exif || opts.all,
onProgress: (msg) => ctx.stderr.write(msg + "\n"),
});
return 0;
return failedMLBatches > 0 ? 1 : 0;
} finally {
await lib.close();
}
+35 -8
View File
@@ -5,7 +5,11 @@ import exifReader from "exif-reader";
import type { Client } from "./client.js";
import type { Library, Photo } from "./library/index.js";
import { sanitizeFileName } from "./filename.js";
import { fetchMLData } from "./mldata-fetch.js";
import {
fetchMLDataBatch,
MLDATA_BATCH_SIZE,
type MLData,
} from "./mldata-fetch.js";
import type { EnteFile } from "./model/types.js";
export type ProgressCallback = (message: string) => void;
@@ -137,13 +141,14 @@ const extractExif = async (
// of plain JSON: account, per-collection, and per-file records including the
// private and public magic metadata and (by default) the ML data. Collections
// and files are enumerated from the library's cache rather than a fresh server
// scan; the ML fetch and EXIF extraction are unchanged.
// scan. Returns how many ML data requests failed; their files are still
// written, with `mlDataError` in place of `mlData`.
export const runMetadataBackup = async (
lib: Library,
client: Client,
outDir: string,
opts?: MetadataBackupOptions,
): Promise<void> => {
): Promise<{ failedMLBatches: number }> => {
const log = opts?.onProgress ?? (() => {});
const wantExif = opts?.exif ?? false;
@@ -208,12 +213,31 @@ export const runMetadataBackup = async (
}
}
// One failed request (retries exhausted) must not end the dump: its files
// get the reason in `mlDataError` and the other batches go on.
log("Fetching ML data (face detections, CLIP embeddings)...");
const mlDataMap = await fetchMLData(
client.getApiClient(),
[...fileKeys.keys()],
fileKeys,
);
const mlDataMap = new Map<number, MLData>();
const mlDataErrors = new Map<number, string>();
let failedMLBatches = 0;
const fileIDs = [...fileKeys.keys()];
for (let i = 0; i < fileIDs.length; i += MLDATA_BATCH_SIZE) {
const batch = fileIDs.slice(i, i + MLDATA_BATCH_SIZE);
try {
const result = await fetchMLDataBatch(
client.getApiClient(),
batch,
fileKeys,
);
for (const [id, payload] of result) mlDataMap.set(id, payload);
} catch (err) {
const reason = err instanceof Error ? err.message : String(err);
failedMLBatches++;
log(
`ML data request for ${batch.length} file(s) failed: ${reason}`,
);
for (const id of batch) mlDataErrors.set(id, reason);
}
}
log(`Got ML data for ${mlDataMap.size} file(s)`);
const writtenFileIDs = new Set<number>();
@@ -233,6 +257,8 @@ export const runMetadataBackup = async (
const ml = mlDataMap.get(file.id);
if (ml) fileMeta.mlData = ml;
const mlError = mlDataErrors.get(file.id);
if (mlError) fileMeta.mlDataError = mlError;
if (wantExif && !writtenFileIDs.has(file.id)) {
log(`[${file.metadata.title}] Extracting EXIF...`);
@@ -253,4 +279,5 @@ export const runMetadataBackup = async (
}
log("Metadata backup complete.");
return { failedMLBatches };
};
+2 -25
View File
@@ -5,9 +5,8 @@
// comes back encrypted under the file's own key and gzipped; decrypting and
// gunzipping yields the JSON payload
// `{ face: { faces: [...] }, clip: { embedding } }`. Ente caps a request at 200
// ids, so `fetchMLData` batches for callers that want many at once while
// `fetchMLDataBatch` is the single-request unit the library submits to its
// request pool.
// ids, so callers that want many at once split them into batches of
// `MLDATA_BATCH_SIZE` and call `fetchMLDataBatch` once per batch.
import { gunzipSync } from "node:zlib";
@@ -69,25 +68,3 @@ export const fetchMLDataBatch = async (
}
return result;
};
// Fetch ML data for arbitrarily many ids, batching at `MLDATA_BATCH_SIZE`. Used
// by the one-shot metadata backup; the library fetches through its request pool
// with `fetchMLDataBatch` instead.
export const fetchMLData = async (
api: ApiClient,
fileIDs: number[],
fileKeys: Map<number, Uint8Array>,
): Promise<Map<number, MLData>> => {
const result = new Map<number, MLData>();
for (let i = 0; i < fileIDs.length; i += MLDATA_BATCH_SIZE) {
const batch = fileIDs.slice(i, i + MLDATA_BATCH_SIZE);
for (const [id, payload] of await fetchMLDataBatch(
api,
batch,
fileKeys,
)) {
result.set(id, payload);
}
}
return result;
};