diff --git a/README.md b/README.md index b98882b..b183973 100644 --- a/README.md +++ b/README.md @@ -471,6 +471,11 @@ accepted for backward compatibility but ignored. `backup-metadata --exif` (alias metadata. The listing and backup commands support `--json` for machine-readable output. +`backup-metadata` fetches ML data in requests of up to 200 files. When a request +still fails after its retries, the error is logged, each of its files is written +with the reason in an `mlDataError` field instead of `mlData`, and the dump goes +on. The exit code is non-zero if any ML data request failed. + `helper fix-missing-thumbnails` regenerates thumbnails for baseline JPEG images only, because the bundled decoder (`jpeg-js`) decodes only JPEG. A non-JPEG image (PNG, HEIC) or a video is reported as `skipped` (unsupported format), kept diff --git a/TODO.md b/TODO.md index 0622278..5bfdace 100644 --- a/TODO.md +++ b/TODO.md @@ -18,6 +18,12 @@ Tag v1.0.0. # Completed Steps +- 2026-09-23: `backup-metadata` no longer stops on one failed ML data request + (issue 101). Each request of up to 200 files is tried on its own; a failed one + is logged, its files are written with the reason in `mlDataError`, and the + command exits 1 once the dump is complete. `fetchMLData`, which only this + command used, is gone; the command calls `fetchMLDataBatch` per batch. + - 2026-09-23: Single-sourced the version string (issue 5). `package.json` is the only place it is written: `src/index.ts` imports it for `VERSION` and `bin/quak.ts` passes `VERSION` to commander. tsc copies `package.json` to diff --git a/src/cli-commands.ts b/src/cli-commands.ts index d72b607..72a9abd 100644 --- a/src/cli-commands.ts +++ b/src/cli-commands.ts @@ -323,11 +323,11 @@ export const backupMetadataCommand = async ( if (!client) return 1; const lib = await openReadLibrary(ctx, client); try { - await runMetadataBackup(lib, client, dir, { + const { failedMLBatches } = await runMetadataBackup(lib, client, dir, { exif: opts.exif || opts.all, onProgress: (msg) => ctx.stderr.write(msg + "\n"), }); - return 0; + return failedMLBatches > 0 ? 1 : 0; } finally { await lib.close(); } diff --git a/src/metadata-backup.ts b/src/metadata-backup.ts index 8497e2d..12435cb 100644 --- a/src/metadata-backup.ts +++ b/src/metadata-backup.ts @@ -5,7 +5,11 @@ import exifReader from "exif-reader"; import type { Client } from "./client.js"; import type { Library, Photo } from "./library/index.js"; import { sanitizeFileName } from "./filename.js"; -import { fetchMLData } from "./mldata-fetch.js"; +import { + fetchMLDataBatch, + MLDATA_BATCH_SIZE, + type MLData, +} from "./mldata-fetch.js"; import type { EnteFile } from "./model/types.js"; export type ProgressCallback = (message: string) => void; @@ -137,13 +141,14 @@ const extractExif = async ( // of plain JSON: account, per-collection, and per-file records including the // private and public magic metadata and (by default) the ML data. Collections // and files are enumerated from the library's cache rather than a fresh server -// scan; the ML fetch and EXIF extraction are unchanged. +// scan. Returns how many ML data requests failed; their files are still +// written, with `mlDataError` in place of `mlData`. export const runMetadataBackup = async ( lib: Library, client: Client, outDir: string, opts?: MetadataBackupOptions, -): Promise => { +): Promise<{ failedMLBatches: number }> => { const log = opts?.onProgress ?? (() => {}); const wantExif = opts?.exif ?? false; @@ -208,12 +213,31 @@ export const runMetadataBackup = async ( } } + // One failed request (retries exhausted) must not end the dump: its files + // get the reason in `mlDataError` and the other batches go on. log("Fetching ML data (face detections, CLIP embeddings)..."); - const mlDataMap = await fetchMLData( - client.getApiClient(), - [...fileKeys.keys()], - fileKeys, - ); + const mlDataMap = new Map(); + const mlDataErrors = new Map(); + let failedMLBatches = 0; + const fileIDs = [...fileKeys.keys()]; + for (let i = 0; i < fileIDs.length; i += MLDATA_BATCH_SIZE) { + const batch = fileIDs.slice(i, i + MLDATA_BATCH_SIZE); + try { + const result = await fetchMLDataBatch( + client.getApiClient(), + batch, + fileKeys, + ); + for (const [id, payload] of result) mlDataMap.set(id, payload); + } catch (err) { + const reason = err instanceof Error ? err.message : String(err); + failedMLBatches++; + log( + `ML data request for ${batch.length} file(s) failed: ${reason}`, + ); + for (const id of batch) mlDataErrors.set(id, reason); + } + } log(`Got ML data for ${mlDataMap.size} file(s)`); const writtenFileIDs = new Set(); @@ -233,6 +257,8 @@ export const runMetadataBackup = async ( const ml = mlDataMap.get(file.id); if (ml) fileMeta.mlData = ml; + const mlError = mlDataErrors.get(file.id); + if (mlError) fileMeta.mlDataError = mlError; if (wantExif && !writtenFileIDs.has(file.id)) { log(`[${file.metadata.title}] Extracting EXIF...`); @@ -253,4 +279,5 @@ export const runMetadataBackup = async ( } log("Metadata backup complete."); + return { failedMLBatches }; }; diff --git a/src/mldata-fetch.ts b/src/mldata-fetch.ts index d1b58b0..e5a254d 100644 --- a/src/mldata-fetch.ts +++ b/src/mldata-fetch.ts @@ -5,9 +5,8 @@ // comes back encrypted under the file's own key and gzipped; decrypting and // gunzipping yields the JSON payload // `{ face: { faces: [...] }, clip: { embedding } }`. Ente caps a request at 200 -// ids, so `fetchMLData` batches for callers that want many at once while -// `fetchMLDataBatch` is the single-request unit the library submits to its -// request pool. +// ids, so callers that want many at once split them into batches of +// `MLDATA_BATCH_SIZE` and call `fetchMLDataBatch` once per batch. import { gunzipSync } from "node:zlib"; @@ -69,25 +68,3 @@ export const fetchMLDataBatch = async ( } return result; }; - -// Fetch ML data for arbitrarily many ids, batching at `MLDATA_BATCH_SIZE`. Used -// by the one-shot metadata backup; the library fetches through its request pool -// with `fetchMLDataBatch` instead. -export const fetchMLData = async ( - api: ApiClient, - fileIDs: number[], - fileKeys: Map, -): Promise> => { - const result = new Map(); - for (let i = 0; i < fileIDs.length; i += MLDATA_BATCH_SIZE) { - const batch = fileIDs.slice(i, i + MLDATA_BATCH_SIZE); - for (const [id, payload] of await fetchMLDataBatch( - api, - batch, - fileKeys, - )) { - result.set(id, payload); - } - } - return result; -}; diff --git a/test/cli/metadata-backup.test.ts b/test/cli/metadata-backup.test.ts index b1f435b..6d597e5 100644 --- a/test/cli/metadata-backup.test.ts +++ b/test/cli/metadata-backup.test.ts @@ -38,7 +38,7 @@ import { join } from "node:path"; import { tmpdir } from "node:os"; import sodium from "libsodium-wrappers-sumo"; import { SRP, SrpServer } from "fast-srp-hap"; -import { beforeAll, afterAll, describe, expect, it } from "vitest"; +import { beforeAll, afterAll, describe, expect, it, vi } from "vitest"; import { init, toBase64, @@ -53,8 +53,16 @@ import { runMetadataBackup, type MetadataBackupOptions, } from "../../src/metadata-backup.js"; +import { backupMetadataCommand } from "../../src/cli-commands.js"; import type { KeyAttributes } from "../../src/auth/types.js"; +// One file per ML data request, so the two files of the mock account are +// fetched in two requests and one of them can fail on its own. +vi.mock("../../src/mldata-fetch.js", async (importOriginal) => ({ + ...(await importOriginal()), + MLDATA_BATCH_SIZE: 1, +})); + const TEST_EMAIL = "metabackup@example.com"; const TEST_PASSWORD = "metapass"; const TEST_OPS = 2; @@ -347,7 +355,8 @@ const buildMetaMock = async (): Promise => { }; }; -const buildMetaFetch = (m: MetaMockState) => { +// `failMLDataFor`: answer 500 to every ML data request that asks for this file. +const buildMetaFetch = (m: MetaMockState, failMLDataFor?: number) => { let srpServer: SrpServer; return (async ( input: RequestInfo | URL, @@ -403,6 +412,8 @@ const buildMetaFetch = (m: MetaMockState) => { } if (path === "/files/data/fetch") { const body = JSON.parse(init?.body as string); + if ((body.fileIDs as number[]).includes(failMLDataFor!)) + return new Response("server error", { status: 500 }); const data = (body.fileIDs as number[]) .filter((id: number) => m.encryptedMLData[id]) .map((id: number) => ({ @@ -636,3 +647,63 @@ describe("quak backup-metadata", () => { expect(failedMeta.imageMetadataError).toEqual(expect.any(String)); }); }); + +describe("quak backup-metadata when an ML data request fails", () => { + // Run the CLI command against the mock and return its exit code, stderr + // and output directory. + const runCommand = async (failMLDataFor?: number) => { + const client = await Client.login({ + email: TEST_EMAIL, + password: TEST_PASSWORD, + apiOptions: { + fetch: buildMetaFetch(mock, failMLDataFor), + retry: { sleep: async () => {} }, + }, + }); + const outDir = mkdtempSync(join(testDir, "ml-fail-")); + let stderr = ""; + const code = await backupMetadataCommand( + { + stdout: { write: () => true }, + stderr: { write: (text: string) => (stderr += text) }, + sessionDir: testDir, + cacheDir: mkdtempSync(join(testDir, "cache-")), + loadSession: () => client, + }, + outDir, + {}, + ); + return { code, stderr, outDir }; + }; + + it("writes every file, marks the failed batch's files, and exits 1", async () => { + const { code, stderr, outDir } = await runCommand(200); + + expect(code).toBe(1); + expect(stderr).toContain("ML data request for 1 file(s) failed"); + + const ok = JSON.parse( + readFileSync( + join(outDir, "collections", "10-Vacation", "100.json"), + "utf-8", + ), + ); + expect(ok.mlData.clip.embedding).toEqual([0.5, 0.6, 0.7]); + expect(ok.mlDataError).toBeUndefined(); + + const failed = JSON.parse( + readFileSync( + join(outDir, "collections", "20-__Work", "200.json"), + "utf-8", + ), + ); + expect(failed.metadata.title).toBe("diagram.png"); + expect(failed.mlData).toBeUndefined(); + expect(failed.mlDataError).toContain("500"); + }); + + it("exits 0 when every ML data request succeeds", async () => { + const { code } = await runCommand(); + expect(code).toBe(0); + }); +});