diff --git a/test/api/client.test.ts b/test/api/client.test.ts index 2d1522c..387da4f 100644 --- a/test/api/client.test.ts +++ b/test/api/client.test.ts @@ -29,12 +29,23 @@ * return a `ReadableStream` from the appropriate CDN * (or the self-hosted fallback path). * + * - Retries and timeouts. Every request is issued under a deadline and, + * where it is safe to do so, retried with exponential backoff. The + * policy is `src/retry.ts`; what the last section of this file + * documents is which requests get it and which deliberately do not. + * * All tests inject a fake `fetch` via the constructor so nothing touches * the network. The fake records every call for assertion. */ import { describe, expect, it } from "vitest"; -import { ApiClient, ApiError } from "../../src/api/client.js"; +import { + ApiClient, + ApiError, + DEFAULT_DOWNLOAD_TIMEOUT_MS, + DEFAULT_REQUEST_TIMEOUT_MS, +} from "../../src/api/client.js"; +import type { RetryOptions } from "../../src/retry.js"; // --------------------------------------------------------------------------- // Test helpers @@ -102,6 +113,92 @@ const recordingFetch = ( return { fetch: fake as typeof globalThis.fetch, calls }; }; +/** + * One scripted outcome for a single `fetch` call: + * + * - a `Response`, returned as-is; + * - an `Error`, thrown — this is how `fetch` reports a network failure; + * - `HANG`, a request that never answers until its own deadline aborts it. + * + * `HANG` is what makes the timeout tests honest. A fake that resolved after a + * delay would be testing the clock; this one resolves *only* when the signal + * quak attached fires. If no signal is attached, or the signal is not wired to + * the body, the promise never settles and the test fails on its own timeout + * rather than passing by accident. + */ +const HANG = Symbol("hang until aborted"); +type FetchStep = Response | Error | typeof HANG; + +const scriptedFetch = ( + ...steps: FetchStep[] +): { + fetch: typeof globalThis.fetch; + calls: { url: string; init: RequestInit | undefined }[]; +} => { + const calls: { url: string; init: RequestInit | undefined }[] = []; + let i = 0; + const fake = async ( + input: RequestInfo | URL, + init?: RequestInit, + ): Promise => { + const url = + typeof input === "string" + ? input + : input instanceof URL + ? input.href + : input.url; + calls.push({ url, init }); + const step = steps[i++]; + if (step === undefined) { + throw new Error(`scriptedFetch: no step for call #${i - 1}`); + } + if (step === HANG) { + return new Promise((_resolve, reject) => { + const signal = init?.signal; + if (!signal) return; + if (signal.aborted) { + reject(signal.reason as Error); + return; + } + signal.addEventListener( + "abort", + () => reject(signal.reason as Error), + { once: true }, + ); + }); + } + if (step instanceof Error) throw step; + return step; + }; + return { fetch: fake as typeof globalThis.fetch, calls }; +}; + +/** An error shaped like a Node transport failure: the errno is on `.code`. */ +const errnoError = (code: string, message = code): Error => + Object.assign(new Error(message), { code }); + +/** + * A retry policy with the waiting removed. Backoff arithmetic is covered in + * `test/retry/retry.test.ts`; what the tests below are about is *how many + * requests* each call site issues, so they inject a `sleep` that returns + * immediately. Nothing in this file waits. + */ +const noWait: RetryOptions = { + sleep: () => Promise.resolve(), + random: () => 0, +}; + +/** Drain a stream and return the bytes, so body-level failures surface. */ +const readAll = async (stream: ReadableStream): Promise => { + const reader = stream.getReader(); + let total = 0; + for (;;) { + const { done, value } = await reader.read(); + if (value) total += value.length; + if (done) return total; + } +}; + // --------------------------------------------------------------------------- // Tests // --------------------------------------------------------------------------- @@ -383,3 +480,445 @@ describe("ApiClient.getFileStream / getThumbnailStream", () => { expect(headers.get("X-Auth-Token")).toBe("tk"); }); }); + +// --------------------------------------------------------------------------- +// Retries +// +// The rule the whole section turns on: a request is repeated only when +// repeating it could produce a different answer, and only when repeating it +// cannot do harm. Those are two separate questions and the second one is why +// `postJSON` and `putJSON` behave differently from everything else here. +// +// Every assertion below counts requests. None of them measures how long +// anything took: the retry policy's `sleep` is injected and returns +// immediately, so a machine under load and an idle one produce identical +// results. +// --------------------------------------------------------------------------- + +describe("ApiClient retries", () => { + it("issues exactly one request for a 404", async () => { + // A 404 is an answer, not a failure to get one. Repeating it wastes + // a round trip and — for `listMissingThumbnails`, which reads a 404 + // as "this thumbnail really is missing" — delays a correct result. + // The script holds five responses; only the first may be consumed. + const { fetch, calls } = scriptedFetch( + textResponse("Not Found", 404), + textResponse("Not Found", 404), + textResponse("Not Found", 404), + textResponse("Not Found", 404), + textResponse("Not Found", 404), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + await expect(client.getJSON("/missing")).rejects.toBeInstanceOf( + ApiError, + ); + expect(calls).toHaveLength(1); + }); + + it("retries a 500 up to the configured attempt count, then throws", async () => { + // `attempts` is a total, not a number of retries: three attempts mean + // three requests. The script offers five responses so that a client + // which ignored the limit would be visible as a count of 4 or 5 + // rather than as a crash. + const { fetch, calls } = scriptedFetch( + textResponse("boom", 500), + textResponse("boom", 500), + textResponse("boom", 500), + textResponse("boom", 500), + textResponse("boom", 500), + ); + const client = new ApiClient({ + fetch, + retry: { ...noWait, attempts: 3 }, + }); + + const err: unknown = await client + .getJSON("/flaky") + .catch((e: unknown) => e); + + expect(err).toBeInstanceOf(ApiError); + expect((err as ApiError).status).toBe(500); + expect(calls).toHaveLength(3); + }); + + it("returns the first successful response after a 503", async () => { + const { fetch, calls } = scriptedFetch( + textResponse("unavailable", 503), + jsonResponse({ ok: true }), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + await expect( + client.getJSON<{ ok: boolean }>("/health"), + ).resolves.toEqual({ ok: true }); + expect(calls).toHaveLength(2); + }); + + it("retries 408 and 429", async () => { + // The two 4xx codes that are about timing rather than about the + // request. Backoff is precisely the right response to both. + for (const status of [408, 429]) { + const { fetch, calls } = scriptedFetch( + textResponse("wait", status), + jsonResponse({ ok: true }), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + await expect( + client.getJSON("/rate-limited"), + ).resolves.toBeDefined(); + expect(calls).toHaveLength(2); + } + }); + + it("retries a fetch rejection and succeeds on a later attempt", async () => { + // A dropped or refused connection is the failure this policy exists + // for: the request never got an answer, so asking again is free of + // consequence and likely to work. + const { fetch, calls } = scriptedFetch( + new TypeError("fetch failed"), + errnoError("ECONNRESET", "read ECONNRESET"), + jsonResponse({ collections: [] }), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + await expect(client.getJSON("/collections/v2")).resolves.toEqual({ + collections: [], + }); + expect(calls).toHaveLength(3); + }); + + it("applies the policy to file and thumbnail streams", async () => { + // Both CDN endpoints go through the same wrapper. A 503 from + // files.ente.io during a large backup is common enough that not + // retrying it would fail files for no reason. + for (const get of ["getFileStream", "getThumbnailStream"] as const) { + const { fetch, calls } = scriptedFetch( + textResponse("unavailable", 503), + streamResponse(new Uint8Array([1, 2, 3])), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + const stream = await client[get](7); + expect(await readAll(stream)).toBe(3); + expect(calls).toHaveLength(2); + } + }); + + it("does not retry when the caller opts out", async () => { + // `{ retry: false }` exists for one caller: the download layer, which + // wraps request *and* body consumption *and* decryption in a single + // retry of its own. Without the opt-out the two budgets would + // multiply — four attempts each becoming sixteen requests for one + // file. + const { fetch, calls } = scriptedFetch( + textResponse("unavailable", 503), + streamResponse(new Uint8Array([1, 2, 3])), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + await expect( + client.getFileStream(7, { retry: false }), + ).rejects.toBeInstanceOf(ApiError); + expect(calls).toHaveLength(1); + }); + + it("exposes its resolved retry policy to the download layer", async () => { + // The download layer runs its own `withRetry` and must run it under + // the same policy the client was configured with, not under the + // library defaults. + const { fetch } = scriptedFetch(jsonResponse({})); + const client = new ApiClient({ + fetch, + retry: { ...noWait, attempts: 9, baseDelayMs: 7, maxDelayMs: 11 }, + }); + + const policy = client.getRetryOptions(); + expect(policy.attempts).toBe(9); + expect(policy.baseDelayMs).toBe(7); + expect(policy.maxDelayMs).toBe(11); + }); +}); + +describe("ApiClient timeouts", () => { + it("ships bounded default deadlines", () => { + // Asserted here so the README and the code cannot drift. Two numbers + // rather than one, because a deadline that is sane for a JSON call is + // nowhere near enough for a multi-gigabyte body, and a deadline long + // enough for that body would let a hung API call stall a backup for + // ten minutes. + expect(DEFAULT_REQUEST_TIMEOUT_MS).toBe(30_000); + expect(DEFAULT_DOWNLOAD_TIMEOUT_MS).toBe(600_000); + }); + + it("attaches an abort signal to every request", async () => { + const { fetch, calls } = recordingFetch( + jsonResponse({}), + jsonResponse({}), + new Response(null, { status: 200 }), + jsonResponse({}), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + await client.getJSON("/a"); + await client.postJSON("/b", {}); + await client.putFile("https://s3.example/x", new Uint8Array([1])); + await client.putJSON("/c", {}); + + for (const call of calls) { + expect(call.init?.signal).toBeInstanceOf(AbortSignal); + } + }); + + it("gives up on a request that never answers, and retries it", async () => { + // Before this policy existed there was no timeout anywhere in quak: a + // CDN connection that accepted the request and then went quiet would + // hang `quak backup` forever. `HANG` reproduces exactly that — the + // fake never answers, so the only thing that can end the call is the + // deadline quak attached. + const { fetch, calls } = scriptedFetch(HANG, HANG, HANG); + const client = new ApiClient({ + fetch, + requestTimeoutMs: 20, + retry: { ...noWait, attempts: 3 }, + }); + + await expect(client.getJSON("/black-hole")).rejects.toThrow(); + // A timeout is retryable, so all three attempts were spent... + expect(calls).toHaveLength(3); + // ...each under its own fresh deadline, not one shared one that + // expired during the first attempt. + const signals = calls.map((c) => c.init?.signal); + expect(new Set(signals).size).toBe(3); + }, 5000); + + it("recovers when a later attempt answers in time", async () => { + const { fetch, calls } = scriptedFetch(HANG, jsonResponse({ ok: 1 })); + const client = new ApiClient({ + fetch, + requestTimeoutMs: 20, + retry: noWait, + }); + + await expect(client.getJSON("/slow-then-fast")).resolves.toEqual({ + ok: 1, + }); + expect(calls).toHaveLength(2); + }, 5000); + + it("aborts a body that stalls after the headers arrived", async () => { + // The failure mode that a naive timeout misses. `getFileStream` + // returns as soon as headers arrive; the bytes are pulled later, in + // the download layer. A deadline that only guarded the initial fetch + // would leave the identical hang one layer down — which is where + // multi-megabyte photo downloads actually stall. + // + // This fake resolves its headers immediately and then serves a body + // that never produces a chunk and never observes the signal, so the + // only thing that can unblock the read is quak's own enforcement of + // the deadline over the stream it hands out. + const stalling = new Response( + new ReadableStream({ + pull: () => new Promise(() => {}), + }), + { status: 200 }, + ); + const { fetch } = scriptedFetch(stalling); + const client = new ApiClient({ + fetch, + downloadTimeoutMs: 20, + retry: { ...noWait, attempts: 1 }, + }); + + const stream = await client.getFileStream(42); + const err: unknown = await readAll(stream).catch((e: unknown) => e); + + expect(err).toBeInstanceOf(Error); + expect((err as Error).name).toBe("TimeoutError"); + }, 5000); + + it("lets a body that arrives in time through untouched", async () => { + // The counterpart to the previous test: enforcing the deadline over + // the stream must not corrupt or truncate a body that is simply being + // read normally. + const payload = new Uint8Array([9, 8, 7, 6, 5]); + const { fetch } = scriptedFetch(streamResponse(payload)); + const client = new ApiClient({ fetch, retry: noWait }); + + const stream = await client.getFileStream(42); + const reader = stream.getReader(); + const chunks: Uint8Array[] = []; + for (;;) { + const { done, value } = await reader.read(); + if (done) break; + chunks.push(value); + } + const joined = new Uint8Array(chunks.reduce((n, c) => n + c.length, 0)); + let offset = 0; + for (const c of chunks) { + joined.set(c, offset); + offset += c.length; + } + expect(joined).toEqual(payload); + }); +}); + +describe("ApiClient error typing", () => { + it("throws ApiError with the status when a presigned PUT fails", async () => { + // `putFile` used to throw a bare Error with the status baked into a + // string. Nothing downstream could classify it, so a 500 from S3 was + // indistinguishable from a bug and could never be retried. + const { fetch, calls } = scriptedFetch( + new Response("Forbidden", { status: 403 }), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + const err: unknown = await client + .putFile("https://s3.example/obj", new Uint8Array([1, 2])) + .catch((e: unknown) => e); + + expect(err).toBeInstanceOf(ApiError); + expect((err as ApiError).status).toBe(403); + expect(calls).toHaveLength(1); + }); + + it("retries a presigned PUT on a 5xx", async () => { + // A presigned PUT writes the whole object at one key in one request, + // so repeating it either overwrites the same bytes or lands them for + // the first time. There is no partial state to protect. + const { fetch, calls } = scriptedFetch( + new Response("slow down", { status: 503 }), + new Response(null, { status: 200 }), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + await client.putFile("https://s3.example/obj", new Uint8Array([1, 2])); + expect(calls).toHaveLength(2); + }); + + it("throws ApiError when a download response has no body", async () => { + // Also previously a bare Error. It carries the response status so a + // caller can see what arrived — and it is *not* retried: a 200 with + // no body is a malformed response, and asking again produces the same + // malformed response. + for (const get of ["getFileStream", "getThumbnailStream"] as const) { + const { fetch, calls } = scriptedFetch( + new Response(null, { status: 200 }), + new Response(null, { status: 200 }), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + const err: unknown = await client[get](5).catch((e: unknown) => e); + + expect(err).toBeInstanceOf(ApiError); + expect((err as ApiError).status).toBe(200); + expect(calls).toHaveLength(1); + } + }); +}); + +describe("ApiClient non-idempotent requests", () => { + /** + * `postJSON` and `putJSON` carry quak's only requests that change server + * state: `/users/srp/create-session`, `/users/two-factor/verify` — which + * consumes one of a small number of 2FA attempts — and `/files/thumbnail`. + * + * They are retried only when the failure proves the request never reached + * the server, which in practice means the connection was never + * established. Everything else is ambiguous: a 5xx proves the server did + * process the request, and a reset or a timeout can arrive after it did. + * Replaying under that ambiguity can burn a 2FA attempt or register a + * thumbnail twice, and neither is worth the round trip it saves. + */ + it("does not replay a POST after a 5xx", async () => { + const { fetch, calls } = scriptedFetch( + textResponse("boom", 500), + jsonResponse({ sessionID: "second" }), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + await expect( + client.postJSON("/users/two-factor/verify", { code: "123456" }), + ).rejects.toBeInstanceOf(ApiError); + expect(calls).toHaveLength(1); + }); + + it("does not replay a POST after a mid-flight connection reset", async () => { + // A reset can happen after the request was fully sent and acted on. + // It is retryable in general — `getJSON` retries it — but it is not + // replay-safe. + const { fetch, calls } = scriptedFetch( + errnoError("ECONNRESET", "socket hang up"), + jsonResponse({ sessionID: "second" }), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + await expect( + client.postJSON("/users/srp/create-session", {}), + ).rejects.toThrow(/ECONNRESET|socket hang up/); + expect(calls).toHaveLength(1); + }); + + it("does not replay a POST after a timeout", async () => { + // A deadline says nothing about whether the server acted. + const { fetch, calls } = scriptedFetch(HANG, jsonResponse({})); + const client = new ApiClient({ + fetch, + requestTimeoutMs: 20, + retry: noWait, + }); + + await expect(client.postJSON("/users/ott", {})).rejects.toThrow(); + expect(calls).toHaveLength(1); + }, 5000); + + it("replays a POST when the connection was never established", async () => { + // A refused connection or a DNS failure happens before any request + // byte is written, so the server cannot have seen it. This is the one + // case where replaying is provably harmless. + const { fetch, calls } = scriptedFetch( + new TypeError("fetch failed", { + cause: errnoError("ECONNREFUSED", "connect ECONNREFUSED"), + }), + errnoError("EAI_AGAIN", "getaddrinfo EAI_AGAIN api.ente.io"), + jsonResponse({ sessionID: "third" }), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + await expect( + client.postJSON<{ sessionID: string }>( + "/users/srp/create-session", + {}, + ), + ).resolves.toEqual({ sessionID: "third" }); + expect(calls).toHaveLength(3); + }); + + it("applies the same rule to PUT", async () => { + // `/files/thumbnail` is reached through `putJSON`. + const failing = scriptedFetch( + textResponse("boom", 503), + jsonResponse({}), + ); + const failingClient = new ApiClient({ + fetch: failing.fetch, + retry: noWait, + }); + await expect( + failingClient.updateThumbnail(1, "key", "header"), + ).rejects.toBeInstanceOf(ApiError); + expect(failing.calls).toHaveLength(1); + + const refused = scriptedFetch( + errnoError("ECONNREFUSED", "connect ECONNREFUSED"), + jsonResponse({}), + ); + const refusedClient = new ApiClient({ + fetch: refused.fetch, + retry: noWait, + }); + await refusedClient.updateThumbnail(1, "key", "header"); + expect(refused.calls).toHaveLength(2); + }); +}); diff --git a/test/cli/backup.test.ts b/test/cli/backup.test.ts index f559b80..5f2efab 100644 --- a/test/cli/backup.test.ts +++ b/test/cli/backup.test.ts @@ -411,11 +411,22 @@ describe("quak backup", () => { // File 101 (sunset.jpg) will return HTTP 500. The other two // files must still download. The result must report the failure // without throwing. + // + // A 500 is retryable, so this file now costs several requests before + // it is given up on — that is the point of the retry policy, and + // `runBackup`'s own resilience is unchanged by it: the retry lives + // strictly below this loop, and an exhausted file is still logged, + // counted, and stepped over rather than aborting the run. The + // injected `sleep` is what keeps the suite from actually waiting out + // the backoff. const outDir = join(testDir, "partial-failure"); const client = await Client.login({ email: TEST_EMAIL, password: TEST_PASSWORD, - apiOptions: { fetch: buildMockFetch(mock, { failFileID: 101 }) }, + apiOptions: { + fetch: buildMockFetch(mock, { failFileID: 101 }), + retry: { sleep: () => Promise.resolve(), random: () => 0 }, + }, }); const result = await runBackup(client, outDir); diff --git a/test/download/download.test.ts b/test/download/download.test.ts index db50a9d..0608381 100644 --- a/test/download/download.test.ts +++ b/test/download/download.test.ts @@ -34,6 +34,15 @@ * forever after. The staging file and the rename are observed directly (see * the `rename` hook below), not inferred from an empty directory. * + * 3. **A failed transfer is retried as a whole.** A download is a request, a + * stream consumption, and a decryption, and only the first of those three + * happens inside `ApiClient`. A socket reset after the response headers + * have arrived therefore surfaces here, in the download layer — and that is + * the dominant failure mode for multi-megabyte photos over a CDN. So the + * entire sequence is retried as one unit, not just the request. The + * secretstream pull state is not resumable and there is no Range support, + * so a retry starts the file over from byte zero. + * * These tests build synthetic encrypted files using sodium's push API, * serve them from a mock fetch, and verify the decrypted output on disk. */ @@ -61,6 +70,8 @@ import { } from "vitest"; import { init, toBase64, STREAM_CHUNK_SIZE } from "../../src/crypto/index.js"; import { ApiClient } from "../../src/api/client.js"; +import { ApiError, TruncatedStreamError } from "../../src/errors.js"; +import type { RetryOptions } from "../../src/retry.js"; import { downloadFile, downloadThumbnail } from "../../src/download/index.js"; import type { EnteFile, FileMetadata } from "../../src/model/types.js"; @@ -326,18 +337,92 @@ beforeAll(() => { multiChunk = encryptMultiChunkBody(multiChunkKey, 1, 1024); }); +/** An error shaped like a Node transport failure: the errno is on `.code`. */ +const errnoError = (code: string, message = code): Error => + Object.assign(new Error(message), { code }); + +/** + * A retry policy with the waiting removed, used by every fixture in this + * file. Backoff arithmetic belongs to `test/retry/retry.test.ts`; here the + * only interesting quantity is how many requests a download issued, so the + * injected `sleep` returns immediately and nothing in this file waits. + */ +const noWait: RetryOptions = { + sleep: () => Promise.resolve(), + random: () => 0, +}; + +/** + * One scripted outcome for a single request to the CDN. + * + * - `body` — a complete response body. + * - `status` — an HTTP error response. + * - `reset` — a response whose headers arrive, whose body delivers `bytes`, + * and which then dies with a socket reset. This is the failure that + * motivates retrying the download rather than the request: by the time it + * happens `ApiClient` has already returned successfully. + */ +type BodyStep = + | { kind: "body"; bytes: Uint8Array } + | { kind: "status"; status: number } + | { kind: "reset"; bytes: Uint8Array }; + +/** + * A fetch that serves one scripted step per call and counts the calls. It + * deliberately refuses to serve more requests than it was given steps for, so + * a retry loop that ran away is a test failure rather than a silent success. + */ +const scriptedCdnFetch = ( + ...steps: BodyStep[] +): { fetch: typeof globalThis.fetch; requests: () => number } => { + let calls = 0; + const fake = async (): Promise => { + const step = steps[calls++]; + if (step === undefined) { + throw new Error(`scriptedCdnFetch: no step for request #${calls}`); + } + if (step.kind === "status") { + return new Response("error", { status: step.status }); + } + if (step.kind === "body") { + return new Response(step.bytes, { status: 200 }); + } + const bytes = step.bytes; + return new Response( + new ReadableStream({ + start(controller) { + controller.enqueue(bytes); + controller.error( + errnoError("ECONNRESET", "aborted by peer"), + ); + }, + }), + { status: 200 }, + ); + }; + return { fetch: fake as typeof globalThis.fetch, requests: () => calls }; +}; + /** * Build an EnteFile plus ApiClient whose file *and* thumbnail streams both * serve `body` under `header`. The download path under test is otherwise * identical for the two, so every truncation/atomicity case below runs * against both entry points from a single fixture. + * + * The default policy here is a single attempt. The failure-contract tests are + * about what the caller and the filesystem are left with, not about how many + * times quak asked; pinning attempts to one keeps them saying exactly that, + * and keeps them from re-decrypting a 4 MiB fixture four times over. The + * retry counts have their own tests at the bottom of this file, which set the + * attempt count explicitly. */ const fixtureFor = ( key: Uint8Array, header: Uint8Array, body: Uint8Array, + retry: RetryOptions = { ...noWait, attempts: 1 }, ): { api: ApiClient; file: EnteFile } => ({ - api: new ApiClient({ fetch: mockFetchForBody(body) }), + api: new ApiClient({ fetch: mockFetchForBody(body), retry }), file: buildMockEnteFile(key, header, header), }); @@ -502,9 +587,17 @@ describe.each(entryPoints)( ); const outPath = join(freshDir(), "truncated.bin"); - await expect(download(api, file, outPath)).rejects.toThrow( - /truncated/i, + const err: unknown = await download(api, file, outPath).catch( + (e: unknown) => e, ); + + // The type, not the wording, is the contract. The retry policy + // classifies truncation as worth another attempt, and it decides + // that with `instanceof`: matching on message text would make + // rewording a diagnostic silently turn every truncated download + // into a permanent failure. + expect(err).toBeInstanceOf(TruncatedStreamError); + expect((err as Error).message).toMatch(/truncated/i); }); it("rejects a body whose final chunk arrived only in part", async () => { @@ -537,7 +630,7 @@ describe.each(entryPoints)( (e: unknown) => e, ); - expect(err).toBeInstanceOf(Error); + expect(err).toBeInstanceOf(TruncatedStreamError); expect((err as Error).message).toMatch(/truncated/i); expect((err as Error).cause).toBeInstanceOf(Error); expect(((err as Error).cause as Error).message).toMatch( @@ -564,8 +657,8 @@ describe.each(entryPoints)( ); const outPath = join(freshDir(), "empty.bin"); - await expect(download(api, file, outPath)).rejects.toThrow( - /truncated/i, + await expect(download(api, file, outPath)).rejects.toBeInstanceOf( + TruncatedStreamError, ); }); @@ -588,8 +681,8 @@ describe.each(entryPoints)( const dir = freshDir(); const outPath = join(dir, "absent.bin"); - await expect(download(api, file, outPath)).rejects.toThrow( - /truncated/i, + await expect(download(api, file, outPath)).rejects.toBeInstanceOf( + TruncatedStreamError, ); expect(existsSync(outPath)).toBe(false); @@ -621,10 +714,18 @@ describe.each(entryPoints)( const dir = freshDir(); const outPath = join(dir, "corrupt.bin"); - await expect(download(api, file, outPath)).rejects.toThrow( - /authentication failed/i, + const err: unknown = await download(api, file, outPath).catch( + (e: unknown) => e, ); + expect(err).toBeInstanceOf(Error); + expect((err as Error).message).toMatch(/authentication failed/i); + // And explicitly *not* the truncation type, because that type is + // what the retry policy keys on: mislabelling corruption as + // truncation would spend the whole attempt budget re-downloading + // a file that will never decrypt. + expect(err).not.toBeInstanceOf(TruncatedStreamError); + expect(existsSync(outPath)).toBe(false); expect(readdirSync(dir)).toEqual([]); }); @@ -648,8 +749,8 @@ describe.each(entryPoints)( const outPath = join(dir, "existing.bin"); writeFileSync(outPath, existing); - await expect(download(api, file, outPath)).rejects.toThrow( - /truncated/i, + await expect(download(api, file, outPath)).rejects.toBeInstanceOf( + TruncatedStreamError, ); expect(readFileSync(outPath)).toEqual(Buffer.from(existing)); @@ -755,3 +856,208 @@ describe.each(entryPoints)( }); }, ); + +// --------------------------------------------------------------------------- +// Retries +// +// What is retried here is the whole download — request, stream consumption, +// decryption — because only the first of those three happens inside +// `ApiClient`. Every assertion counts requests; none of them measures time. +// --------------------------------------------------------------------------- + +describe.each(entryPoints)("$name retries", ({ name, download }) => { + const freshDir = (): string => mkdtempSync(join(testDir, `${name}-retry-`)); + + /** A cheap single-chunk fixture: no 4 MiB encryption in the retry tests. */ + const smallFixture = ( + seed: number, + ): { + key: Uint8Array; + header: Uint8Array; + ciphertext: Uint8Array; + plaintext: Uint8Array; + } => { + const plaintext = patternBytes(1024, seed); + const key = sodium.crypto_secretstream_xchacha20poly1305_keygen(); + const { header, ciphertext } = encryptFileBody(plaintext, key); + return { key, header, ciphertext, plaintext }; + }; + + const clientFor = ( + fetch: typeof globalThis.fetch, + attempts: number, + ): ApiClient => new ApiClient({ fetch, retry: { ...noWait, attempts } }); + + it("retries a connection reset that happened mid-body", async () => { + // The case `ApiClient` cannot see. Its own request succeeded: headers + // arrived, a `ReadableStream` was handed back, and only then did the + // socket die. Retrying the fetch alone would have caught nothing, + // which is why the retry wraps the whole sequence. + const { key, header, ciphertext, plaintext } = smallFixture(41); + const { fetch, requests } = scriptedCdnFetch( + { kind: "reset", bytes: ciphertext.slice(0, 16) }, + { kind: "body", bytes: ciphertext }, + ); + const api = clientFor(fetch, 4); + const file = buildMockEnteFile(key, header, header); + const dir = freshDir(); + const outPath = join(dir, "reset-then-ok.bin"); + + const result = await download(api, file, outPath); + + expect(requests()).toBe(2); + expect(result.bytesWritten).toBe(plaintext.length); + expectSameBytes(readFileSync(outPath), plaintext); + }); + + it("stages one temp file for the attempt that succeeded, not one per attempt", async () => { + // The atomic write stays outside the retry loop. A retried download + // must not leave a trail of half-written scratch files, and the + // destination must be touched exactly once — by the attempt that + // produced a complete, authenticated plaintext. + const { key, header, ciphertext } = smallFixture(42); + const { fetch } = scriptedCdnFetch( + { kind: "reset", bytes: ciphertext.slice(0, 16) }, + { kind: "reset", bytes: ciphertext.slice(0, 16) }, + { kind: "body", bytes: ciphertext }, + ); + const api = clientFor(fetch, 4); + const file = buildMockEnteFile(key, header, header); + const dir = freshDir(); + const outPath = join(dir, "one-stage.bin"); + + await download(api, file, outPath); + + expect(renameHook.calls).toHaveLength(1); + expect(renameHook.calls[0]!.to).toBe(outPath); + expect(readdirSync(dir)).toEqual(["one-stage.bin"]); + }); + + it("retries a truncated body and gives up after the configured attempts", async () => { + // Truncation is retryable — the file on the server is intact, the + // transfer was not — but it is not retryable forever. Three attempts + // configured, three requests, then the caller gets the error. + const key = sodium.crypto_secretstream_xchacha20poly1305_keygen(); + const { header, ciphertext } = encryptNonFinalBody( + patternBytes(256, 43), + key, + ); + const { fetch, requests } = scriptedCdnFetch( + { kind: "body", bytes: ciphertext }, + { kind: "body", bytes: ciphertext }, + { kind: "body", bytes: ciphertext }, + { kind: "body", bytes: ciphertext }, + ); + const api = clientFor(fetch, 3); + const file = buildMockEnteFile(key, header, header); + const dir = freshDir(); + const outPath = join(dir, "always-truncated.bin"); + + await expect(download(api, file, outPath)).rejects.toBeInstanceOf( + TruncatedStreamError, + ); + + expect(requests()).toBe(3); + // Every attempt failed before anything was written, so the directory + // is still empty. + expect(readdirSync(dir)).toEqual([]); + }); + + it("issues exactly one request when the file is gone", async () => { + // A 404 from the CDN is an answer. `runBackup` logs it and moves on; + // spending three more requests and three backoff waits on it would + // slow a large backup down for nothing. + const { key, header } = smallFixture(44); + const { fetch, requests } = scriptedCdnFetch( + { kind: "status", status: 404 }, + { kind: "status", status: 404 }, + { kind: "status", status: 404 }, + { kind: "status", status: 404 }, + ); + const api = clientFor(fetch, 4); + const file = buildMockEnteFile(key, header, header); + const outPath = join(freshDir(), "gone.bin"); + + const err: unknown = await download(api, file, outPath).catch( + (e: unknown) => e, + ); + + expect(err).toBeInstanceOf(ApiError); + expect((err as ApiError).status).toBe(404); + expect(requests()).toBe(1); + }); + + it("retries a 503 from the CDN", async () => { + const { key, header, ciphertext, plaintext } = smallFixture(45); + const { fetch, requests } = scriptedCdnFetch( + { kind: "status", status: 503 }, + { kind: "status", status: 503 }, + { kind: "body", bytes: ciphertext }, + ); + const api = clientFor(fetch, 4); + const file = buildMockEnteFile(key, header, header); + const outPath = join(freshDir(), "flaky-cdn.bin"); + + await download(api, file, outPath); + + expect(requests()).toBe(3); + expectSameBytes(readFileSync(outPath), plaintext); + }); + + it("spends one attempt budget, not one per layer", async () => { + // `ApiClient.getFileStream` retries on its own for direct callers. + // The download layer opts out of that and runs its own retry over the + // whole sequence. If it did not, the two budgets would compose: three + // attempts here would become nine requests to the CDN for a single + // file, and the default four would become sixteen. + const { key, header } = smallFixture(46); + const steps: BodyStep[] = Array.from({ length: 12 }, () => ({ + kind: "status" as const, + status: 503, + })); + const { fetch, requests } = scriptedCdnFetch(...steps); + const api = clientFor(fetch, 3); + const file = buildMockEnteFile(key, header, header); + const outPath = join(freshDir(), "budget.bin"); + + await expect(download(api, file, outPath)).rejects.toBeInstanceOf( + ApiError, + ); + + expect(requests()).toBe(3); + }); +}); + +describe("download retries: corruption is not retried", () => { + it("gives up immediately on a chunk that failed to authenticate", async () => { + // A whole chunk that failed to authenticate while the stream + // continued past it is corruption or a wrong key. Neither is fixed by + // asking again, and a backup run that retried every such file would + // multiply the cost of a genuinely broken file by the attempt count. + // + // This is also the boundary of the single-chunk ambiguity documented + // at the classifier: the split is only achievable because this body + // has more than one chunk. + const corrupted = Uint8Array.from(multiChunk.body); + corrupted[10] ^= 0xff; + const { fetch, requests } = scriptedCdnFetch( + { kind: "body", bytes: corrupted }, + { kind: "body", bytes: corrupted }, + { kind: "body", bytes: corrupted }, + { kind: "body", bytes: corrupted }, + ); + const api = new ApiClient({ fetch, retry: { ...noWait, attempts: 4 } }); + const file = buildMockEnteFile( + multiChunkKey, + multiChunk.header, + multiChunk.header, + ); + const outPath = join(mkdtempSync(join(testDir, "corrupt-")), "c.bin"); + + await expect(downloadFile(api, file, outPath)).rejects.toThrow( + /authentication failed/i, + ); + + expect(requests()).toBe(1); + }); +}); diff --git a/test/retry/retry.test.ts b/test/retry/retry.test.ts new file mode 100644 index 0000000..42eefed --- /dev/null +++ b/test/retry/retry.test.ts @@ -0,0 +1,557 @@ +/** + * Tests for `src/retry.ts` — the retry policy shared by every network + * operation in quak. + * + * Two things live in that module and they are deliberately separate: + * + * - **`isRetryable(err)`**, a pure classifier. Given an error, is trying + * again capable of producing a different answer? Nothing else about the + * error matters: not how it was logged, not where it came from. + * + * - **`withRetry(fn, opts)`**, the loop. It calls `fn`, and while the error + * is classified retryable and attempts remain, it sleeps and calls `fn` + * again. It never inspects errors itself. + * + * The classifier's default answer is *no*. quak is a backup tool: a wrongly + * retried permanent failure costs a user round trips and delays the rest of + * the run, while a wrongly rejected transient failure costs one file that the + * next run picks up. When in doubt, fail fast. + * + * ## Reading the backoff assertions + * + * `withRetry` takes its `sleep` and its `random` as injected functions. Every + * test here passes a `sleep` that records the delay it was asked for and + * returns immediately, so the suite never waits, and a `random` that returns a + * fixed number, so jitter is exact rather than approximate. **No assertion in + * this file (or anywhere else in the suite) is about elapsed wall-clock time.** + * They are about what `withRetry` asked for, and how many times `fn` ran. + * + * The delay before retry number *n* (1-based) is: + * + * random() * min(maxDelayMs, baseDelayMs * 2 ** (n - 1)) + * + * That is exponential backoff with full jitter: the exponential term is the + * *ceiling*, and the actual wait is drawn uniformly below it. Full jitter, + * rather than a fixed delay plus noise, is what stops a client that lost a + * hundred parallel downloads to one CDN blip from re-sending all hundred at + * the same instant. + */ + +import { describe, expect, it } from "vitest"; +import { + DEFAULT_RETRY_OPTIONS, + isRetryable, + isSafeToReplay, + resolveRetryOptions, + withRetry, +} from "../../src/retry.js"; +import { ApiError, TruncatedStreamError } from "../../src/errors.js"; +import { ApiError as ApiErrorFromClient } from "../../src/api/client.js"; + +// --------------------------------------------------------------------------- +// Test helpers +// --------------------------------------------------------------------------- + +/** + * A `sleep` that records what it was asked to wait for and returns + * immediately. This is the whole reason `withRetry` takes an injected sleep: + * the retry policy is exercised in full — every branch, every delay — without + * the suite spending a single millisecond waiting. + */ +const recordingSleep = (): { + sleep: (ms: number) => Promise; + delays: number[]; +} => { + const delays: number[] = []; + return { + sleep: (ms: number): Promise => { + delays.push(ms); + return Promise.resolve(); + }, + delays, + }; +}; + +/** An error shaped like a Node transport failure: the errno is on `.code`. */ +const errnoError = (code: string, message = code): Error => + Object.assign(new Error(message), { code }); + +// --------------------------------------------------------------------------- +// Classification +// --------------------------------------------------------------------------- + +describe("isRetryable: HTTP status codes", () => { + it("does not retry ordinary 4xx responses", () => { + // A 4xx is the server saying the request itself is wrong. Repeating + // it verbatim produces the same answer, so retrying only delays the + // failure the caller has to handle. 404 is the load-bearing case: + // `listMissingThumbnails` depends on a 404 arriving promptly and + // exactly once. + for (const status of [400, 401, 403, 404, 409, 410, 422]) { + expect(isRetryable(new ApiError(`HTTP ${status}`, status))).toBe( + false, + ); + } + }); + + it("retries 408 and 429", () => { + // The two 4xx codes that are statements about timing rather than + // about the request. 408 is the server admitting it gave up waiting; + // 429 is it asking for less traffic — which backoff supplies. + expect(isRetryable(new ApiError("timeout", 408))).toBe(true); + expect(isRetryable(new ApiError("slow down", 429))).toBe(true); + }); + + it("retries every 5xx response", () => { + // A 5xx is the server failing, not the request being wrong. Ente's + // CDN in particular returns 500 and 503 under load. + for (const status of [500, 502, 503, 504, 599]) { + expect(isRetryable(new ApiError(`HTTP ${status}`, status))).toBe( + true, + ); + } + }); + + it("treats a 2xx or 3xx ApiError as not retryable", () => { + // These exist: `getFileStream` raises an ApiError carrying the + // response status when a 200 arrives with a null body. That is a + // malformed response, not a transport failure, and repeating the + // request will produce the same malformed response. + expect(isRetryable(new ApiError("response body is null", 200))).toBe( + false, + ); + expect(isRetryable(new ApiError("redirect", 304))).toBe(false); + }); + + it("uses the same ApiError class that ApiClient exports", () => { + // `ApiError` lives in `src/errors.ts` and is re-exported from + // `src/api/client.ts`, which is where every existing caller and test + // imports it from. If those ever became two separate classes the + // classifier would silently stop recognising errors raised by the + // client, and every 5xx in the wild would be treated as permanent. + expect(ApiErrorFromClient).toBe(ApiError); + expect(new ApiErrorFromClient("boom", 503)).toBeInstanceOf(ApiError); + expect(isRetryable(new ApiErrorFromClient("boom", 503))).toBe(true); + }); +}); + +describe("isRetryable: transport failures", () => { + it("retries a TypeError, which is how fetch reports a failed request", () => { + // Node's fetch rejects with `TypeError: fetch failed` for everything + // below HTTP: DNS failure, refused connection, TLS error, reset + // socket. The real diagnosis is on `cause`, but there is nothing on + // the object that distinguishes it from a TypeError thrown by a bug, + // so this rule is deliberately literal. The cost of the imprecision + // is bounded by the attempt count; the alternative — demanding a + // recognised `cause` — would classify real network failures as + // permanent and fail backups that should have succeeded. + expect(isRetryable(new TypeError("fetch failed"))).toBe(true); + }); + + it("retries an errno carried on the error itself", () => { + for (const code of [ + "ECONNRESET", + "ETIMEDOUT", + "EPIPE", + "ENOTFOUND", + "EAI_AGAIN", + "ECONNREFUSED", + "EHOSTUNREACH", + "ENETUNREACH", + ]) { + expect(isRetryable(errnoError(code))).toBe(true); + } + }); + + it("retries an errno buried in the cause chain", () => { + // undici does not put the errno on the error it throws; it hangs the + // underlying socket error off `cause`, sometimes more than one level + // down. A classifier that only looked at the top-level error would + // see a bare `Error` and call every dropped connection permanent. + const nested = new Error("request to files.ente.io failed", { + cause: new Error("socket hang up", { + cause: errnoError("ECONNRESET", "read ECONNRESET"), + }), + }); + expect(isRetryable(nested)).toBe(true); + }); + + it("accepts a plain object as a cause", () => { + // Not everything in a cause chain is an Error instance. + expect( + isRetryable(new Error("failed", { cause: { code: "ETIMEDOUT" } })), + ).toBe(true); + }); + + it("retries an aborted request", () => { + // `AbortSignal.timeout()` aborts with a `TimeoutError`; an explicit + // `abort()` produces an `AbortError`. quak only ever aborts a request + // on its own deadline, so both mean "this attempt ran out of time", + // which is exactly the condition a later attempt might not hit. + expect(isRetryable(new DOMException("timed out", "TimeoutError"))).toBe( + true, + ); + expect(isRetryable(new DOMException("aborted", "AbortError"))).toBe( + true, + ); + }); + + it("does not confuse an unrelated errno with a transport failure", () => { + // A filesystem error surfaces the same way an errno network error + // does. Retrying a full disk or a missing directory is pointless. + expect(isRetryable(errnoError("ENOSPC", "no space left"))).toBe(false); + expect(isRetryable(errnoError("ENOENT", "no such file"))).toBe(false); + expect(isRetryable(errnoError("EACCES", "permission denied"))).toBe( + false, + ); + }); + + it("terminates on a cause chain that points at itself", () => { + // Defensive: `cause` is an arbitrary user-settable property and + // nothing stops it forming a cycle. Without a bound on the walk this + // classifier would hang the process, which is a worse failure than + // any misclassification. + const looped: Error & { cause?: unknown } = new Error("loop"); + looped.cause = looped; + expect(isRetryable(looped)).toBe(false); + }); +}); + +describe("isRetryable: stream truncation versus corruption", () => { + it("retries a truncated stream", () => { + // Truncation is a transfer that stopped early. The bytes that did + // arrive are useless, but the file on the server is fine, so asking + // again is exactly right. + expect(isRetryable(new TruncatedStreamError("stream truncated"))).toBe( + true, + ); + }); + + it("retries a truncated stream whose cause is an authentication failure", () => { + // The single-chunk ambiguity, recorded on issue #2 and inherited from + // the truncation work: when a body ends part-way through a chunk, + // Poly1305 fails and carries no framing signal, so a cut connection + // and genuinely corrupt bytes are indistinguishable. That case is + // reported as truncation with the authentication failure preserved as + // `cause`, and it is therefore retried. + // + // Retrying is the deliberate choice. For a multi-chunk body the split + // is real — a corrupt chunk mid-stream stays an authentication + // failure, see the next test — but for a single-chunk body (most + // thumbnails, every small file) a wrong key and a cut connection look + // identical. The cost of guessing wrong is bounded: a few extra round + // trips before the same failure. The cost of guessing the other way + // is a silently truncated file kept forever. + const err = new TruncatedStreamError("stream truncated", { + cause: new Error("secretstream chunk authentication failed"), + }); + expect(isRetryable(err)).toBe(true); + }); + + it("does not retry an authentication failure that is not truncation", () => { + // A whole chunk that failed to authenticate while the stream carried + // on past it cannot be a short transfer. It is corruption or a wrong + // key, and no number of retries fixes either. + expect( + isRetryable(new Error("secretstream chunk authentication failed")), + ).toBe(false); + }); +}); + +describe("isRetryable: everything else", () => { + it("does not retry programming errors or unknown values", () => { + expect(isRetryable(new Error("boom"))).toBe(false); + expect(isRetryable(new RangeError("out of range"))).toBe(false); + expect(isRetryable(new SyntaxError("bad JSON"))).toBe(false); + expect(isRetryable("a string")).toBe(false); + expect(isRetryable(undefined)).toBe(false); + expect(isRetryable(null)).toBe(false); + }); +}); + +// --------------------------------------------------------------------------- +// The replay-safety classifier for non-idempotent requests +// --------------------------------------------------------------------------- + +describe("isSafeToReplay", () => { + /** + * `isRetryable` answers "could a retry succeed?". For a POST or a PUT + * that is not the whole question: the other half is "could the first + * attempt already have taken effect on the server?". + * + * quak's non-idempotent calls are `/users/srp/create-session`, + * `/users/two-factor/verify` (which consumes one of a limited number of + * 2FA attempts) and `/files/thumbnail`. A blind replay of any of them can + * do real damage, so they retry only on failures that prove no request + * byte ever reached the server — which means the connection was never + * established. + */ + it("replays only failures where the connection was never established", () => { + for (const code of [ + "ENOTFOUND", + "EAI_AGAIN", + "ECONNREFUSED", + "EHOSTUNREACH", + "ENETUNREACH", + ]) { + expect(isSafeToReplay(errnoError(code))).toBe(true); + } + // Also when undici has buried it, which is how it actually arrives. + expect( + isSafeToReplay( + new TypeError("fetch failed", { + cause: errnoError("ECONNREFUSED"), + }), + ), + ).toBe(true); + }); + + it("does not replay a failure that could have happened after the server acted", () => { + // Every one of these is ambiguous about whether the server processed + // the request. A 5xx proves it did. A reset or a broken pipe can + // arrive after the request was fully sent and handled. A timeout says + // nothing at all about the server's state. A bare `fetch failed` with + // no recognisable cause could be any of them. + expect(isSafeToReplay(new ApiError("HTTP 500", 500))).toBe(false); + expect(isSafeToReplay(new ApiError("HTTP 429", 429))).toBe(false); + expect(isSafeToReplay(errnoError("ECONNRESET"))).toBe(false); + expect(isSafeToReplay(errnoError("EPIPE"))).toBe(false); + expect(isSafeToReplay(errnoError("ETIMEDOUT"))).toBe(false); + expect( + isSafeToReplay(new DOMException("timed out", "TimeoutError")), + ).toBe(false); + expect(isSafeToReplay(new TypeError("fetch failed"))).toBe(false); + }); +}); + +// --------------------------------------------------------------------------- +// The retry loop +// --------------------------------------------------------------------------- + +describe("withRetry", () => { + it("calls the function once and does not sleep when it succeeds", async () => { + const { sleep, delays } = recordingSleep(); + let calls = 0; + const result = await withRetry( + () => { + calls++; + return Promise.resolve("ok"); + }, + { sleep }, + ); + + expect(result).toBe("ok"); + expect(calls).toBe(1); + expect(delays).toEqual([]); + }); + + it("stops at the first success and returns its value", async () => { + const { sleep, delays } = recordingSleep(); + let calls = 0; + const result = await withRetry( + () => { + calls++; + if (calls < 3) { + return Promise.reject(new ApiError("HTTP 503", 503)); + } + return Promise.resolve(calls); + }, + { attempts: 5, sleep }, + ); + + expect(result).toBe(3); + expect(calls).toBe(3); + // Two failures, so two waits — and none after the attempt that + // succeeded. + expect(delays).toHaveLength(2); + }); + + it("gives up after `attempts` calls and throws the last error", async () => { + // `attempts` counts calls, not retries: `attempts: 3` means the + // function runs three times in total. The error that escapes is the + // one from the final attempt, because that is the current state of + // the world; a caller logging it is logging what is true now. + const { sleep, delays } = recordingSleep(); + let calls = 0; + const failure = withRetry( + () => { + calls++; + return Promise.reject(new ApiError(`attempt ${calls}`, 503)); + }, + { attempts: 3, sleep }, + ); + + await expect(failure).rejects.toThrow("attempt 3"); + expect(calls).toBe(3); + // Three attempts, two gaps between them. Sleeping after the last + // attempt would delay the caller's failure for nothing. + expect(delays).toHaveLength(2); + }); + + it("does not retry at all when attempts is 1", async () => { + const { sleep, delays } = recordingSleep(); + let calls = 0; + await expect( + withRetry( + () => { + calls++; + return Promise.reject(new ApiError("HTTP 500", 500)); + }, + { attempts: 1, sleep }, + ), + ).rejects.toThrow("HTTP 500"); + + expect(calls).toBe(1); + expect(delays).toEqual([]); + }); + + it("rethrows a non-retryable error immediately", async () => { + const { sleep, delays } = recordingSleep(); + let calls = 0; + await expect( + withRetry( + () => { + calls++; + return Promise.reject(new ApiError("HTTP 404", 404)); + }, + { attempts: 5, sleep }, + ), + ).rejects.toBeInstanceOf(ApiError); + + expect(calls).toBe(1); + expect(delays).toEqual([]); + }); + + it("preserves the error object, not just its message", async () => { + // Callers classify what escapes: `listMissingThumbnails` needs the + // `ApiError` and its status to tell a genuine 404 from a transient + // failure. Wrapping the error in a "retries exhausted" error would + // break that. + const original = new ApiError("gone", 410, { code: "GONE" }); + const err: unknown = await withRetry(() => Promise.reject(original), { + sleep: () => Promise.resolve(), + }).catch((e: unknown) => e); + + expect(err).toBe(original); + }); + + it("honours a caller-supplied classifier", async () => { + // This is how the non-idempotent call sites narrow the policy: same + // loop, same backoff, stricter question. + const { sleep } = recordingSleep(); + let calls = 0; + await expect( + withRetry( + () => { + calls++; + // Retryable under the default policy... + return Promise.reject(new ApiError("HTTP 503", 503)); + }, + { attempts: 4, sleep, isRetryable: isSafeToReplay }, + ), + ).rejects.toThrow("HTTP 503"); + + // ...but not under `isSafeToReplay`, so it ran exactly once. + expect(calls).toBe(1); + }); +}); + +describe("withRetry backoff", () => { + it("doubles the ceiling on each retry and caps it", async () => { + // `random: () => 1` pins the jitter to the top of its range, which + // makes the ceiling itself observable. The sequence is + // base, base*2, base*4, ... clamped at maxDelayMs — so a long outage + // settles into a steady poll instead of growing to hours. + const { sleep, delays } = recordingSleep(); + await expect( + withRetry(() => Promise.reject(new ApiError("HTTP 500", 500)), { + attempts: 6, + baseDelayMs: 100, + maxDelayMs: 250, + sleep, + random: () => 1, + }), + ).rejects.toThrow(); + + expect(delays).toEqual([100, 200, 250, 250, 250]); + }); + + it("draws each delay uniformly below its ceiling", async () => { + // Full jitter. The exponential value is the maximum wait, not the + // wait itself, so a fleet of clients that failed together does not + // come back in lockstep. + const { sleep, delays } = recordingSleep(); + await expect( + withRetry(() => Promise.reject(new ApiError("HTTP 500", 500)), { + attempts: 4, + baseDelayMs: 100, + maxDelayMs: 10_000, + sleep, + random: () => 0.25, + }), + ).rejects.toThrow(); + + expect(delays).toEqual([25, 50, 100]); + }); + + it("never asks to sleep longer than the cap or less than zero", async () => { + // Whatever `random` returns from its [0, 1) contract, the delay stays + // inside the configured envelope. + const draws = [0, 0.999_999, 0.5, 0.1, 0.9]; + let i = 0; + const { sleep, delays } = recordingSleep(); + await expect( + withRetry(() => Promise.reject(new ApiError("HTTP 500", 500)), { + attempts: 6, + baseDelayMs: 1000, + maxDelayMs: 2000, + sleep, + random: () => draws[i++]!, + }), + ).rejects.toThrow(); + + expect(delays).toHaveLength(5); + for (const d of delays) { + expect(d).toBeGreaterThanOrEqual(0); + expect(d).toBeLessThanOrEqual(2000); + } + expect(delays[0]).toBe(0); + }); +}); + +describe("retry defaults", () => { + it("ships a bounded, documented default policy", () => { + // These are the numbers the README documents. They are asserted here + // so the README and the code cannot drift apart silently: four + // attempts, half a second of base delay, ten seconds of ceiling — + // under 15 seconds of waiting in the worst case, which keeps a + // several-thousand-file backup moving past a bad file rather than + // stalling on it. + expect(DEFAULT_RETRY_OPTIONS.attempts).toBe(4); + expect(DEFAULT_RETRY_OPTIONS.baseDelayMs).toBe(500); + expect(DEFAULT_RETRY_OPTIONS.maxDelayMs).toBe(10_000); + }); + + it("fills in only the fields the caller left out", () => { + const resolved = resolveRetryOptions({ attempts: 2 }); + expect(resolved.attempts).toBe(2); + expect(resolved.baseDelayMs).toBe(DEFAULT_RETRY_OPTIONS.baseDelayMs); + expect(resolved.maxDelayMs).toBe(DEFAULT_RETRY_OPTIONS.maxDelayMs); + expect(typeof resolved.sleep).toBe("function"); + expect(typeof resolved.random).toBe("function"); + }); + + it("resolves to the defaults when given nothing", () => { + expect(resolveRetryOptions()).toEqual(DEFAULT_RETRY_OPTIONS); + expect(resolveRetryOptions({})).toEqual(DEFAULT_RETRY_OPTIONS); + }); + + it("defaults random to a real generator in [0, 1)", () => { + const { random } = resolveRetryOptions(); + for (let i = 0; i < 100; i++) { + const r = random(); + expect(r).toBeGreaterThanOrEqual(0); + expect(r).toBeLessThan(1); + } + }); +}); diff --git a/test/thumbnails/thumbnails.test.ts b/test/thumbnails/thumbnails.test.ts index 314eb3d..2743d17 100644 --- a/test/thumbnails/thumbnails.test.ts +++ b/test/thumbnails/thumbnails.test.ts @@ -32,6 +32,7 @@ import { fixMissingThumbnails, } from "../../src/thumbnails.js"; import type { KeyAttributes } from "../../src/auth/types.js"; +import type { RetryOptions } from "../../src/retry.js"; // --------------------------------------------------------------------------- // Mock server with controllable thumbnail behavior @@ -51,7 +52,7 @@ interface ThumbMockState { filesByCollection: Record[]>; fileCiphertexts: Record; fileKeys: Record; - thumbnailBehavior: Record; + thumbnailBehavior: Record; // Captures from fix operations uploadedThumbnails: { fileID: number; @@ -302,6 +303,9 @@ const buildThumbFetch = (m: ThumbMockState) => { if (behavior === "empty") { return new Response(new Uint8Array(0), { status: 200 }); } + if (behavior === "500") { + return new Response("Internal Server Error", { status: 500 }); + } return new Response("not found", { status: 404 }); } @@ -345,6 +349,38 @@ const buildThumbFetch = (m: ThumbMockState) => { }) as typeof globalThis.fetch; }; +/** + * A retry policy with the waiting removed. `listMissingThumbnails` walks every + * file in the account, so a transient failure is retried; without an injected + * `sleep` these tests would spend real seconds waiting out backoff. + */ +const noWait: RetryOptions = { + sleep: () => Promise.resolve(), + random: () => 0, +}; + +/** Wrap a fetch so the tests can count how often one endpoint was hit. */ +const countingFetch = ( + inner: typeof globalThis.fetch, + match: (url: string) => boolean, +): { fetch: typeof globalThis.fetch; matched: () => number } => { + let matched = 0; + const fake = async ( + input: RequestInfo | URL, + init?: RequestInit, + ): Promise => { + const url = + typeof input === "string" + ? input + : input instanceof URL + ? input.href + : input.url; + if (match(url)) matched++; + return inner(input, init); + }; + return { fetch: fake as typeof globalThis.fetch, matched: () => matched }; +}; + // --------------------------------------------------------------------------- // Tests // --------------------------------------------------------------------------- @@ -378,7 +414,77 @@ describe("listMissingThumbnails", () => { expect(emptyEntry.collection).toBe("Photos"); const notFoundEntry = missing.find((m) => m.fileID === 102)!; - expect(notFoundEntry.reason).toContain("fetch failed"); + // A 404 is the server stating the thumbnail is not there. That is the + // only network answer that means "missing", and the reason says so + // rather than the older catch-all "fetch failed" — which used to + // cover a 500 and a dropped connection too. + expect(notFoundEntry.reason).toContain("not found"); + }); + + it("does not report a thumbnail as missing when the server is failing", async () => { + // The distinction that matters for `helper fix-missing-thumbnails`. + // Reporting a file here leads to downloading the original, + // regenerating a thumbnail, and uploading it over a thumbnail that + // was fine all along — because the server was briefly returning 500s. + // + // File 102 serves 500 on every attempt, so the retries are genuinely + // exhausted. It must still not be reported. + const failingMock = await buildThumbMock(); + failingMock.thumbnailBehavior[102] = "500"; + + const counted = countingFetch( + buildThumbFetch(failingMock), + (url) => url.includes("thumbnails.ente.io") && url.includes("102"), + ); + const client = await Client.login({ + email: TEST_EMAIL, + password: TEST_PASSWORD, + apiOptions: { fetch: counted.fetch, retry: { ...noWait } }, + }); + + const missing = await listMissingThumbnails(client); + + // Only the genuinely empty thumbnail is reported. + expect(missing.map((m) => m.fileID)).toEqual([101]); + // And the 500 was retried rather than accepted as an answer: four + // attempts is the library default. + expect(counted.matched()).toBe(4); + }); + + it("does not report a thumbnail as missing when the connection fails", async () => { + // Same rule for a transport failure, which carries no status at all. + const failingMock = await buildThumbMock(); + const inner = buildThumbFetch(failingMock); + let thumbRequests = 0; + const fetch = (async ( + input: RequestInfo | URL, + init?: RequestInit, + ): Promise => { + const url = + typeof input === "string" + ? input + : input instanceof URL + ? input.href + : input.url; + if (url.includes("thumbnails.ente.io") && url.includes("102")) { + thumbRequests++; + throw Object.assign(new Error("socket hang up"), { + code: "ECONNRESET", + }); + } + return inner(input, init); + }) as typeof globalThis.fetch; + + const client = await Client.login({ + email: TEST_EMAIL, + password: TEST_PASSWORD, + apiOptions: { fetch, retry: { ...noWait } }, + }); + + const missing = await listMissingThumbnails(client); + + expect(missing.map((m) => m.fileID)).toEqual([101]); + expect(thumbRequests).toBe(4); }); it("deduplicates files seen in multiple collections", async () => {