Files
quak/src/retry.ts
T
clawbot d545dcd8b1
check / check (push) Successful in 42s
Harden the retry classifier and pin per-attempt deadlines (closes #80)
A POST or PUT is replayed only when every errno in the cause chain is a
connect errno and the walk reached the end of the chain, and
postJSON/putJSON no longer follow redirects, so a redirect is an
ApiError that is not retried. getRetryOptions() returns a copy. New
tests pin every errno the classifier names, the cause-chain depth
limit, a chain deeper than the limit, a two-error cycle, and a fresh
deadline per attempt for every retrying entry point. The README
endpoint list is now the one place naming the requests the replay rule
covers; code comments point to it.

Model: opus-5-5
2026-09-23 02:38:01 +02:00

215 lines
9.9 KiB
TypeScript

// The retry policy shared by every network operation in quak.
//
// Two independent pieces: a classifier that decides whether an error is worth
// another attempt, and a loop that acts on that decision with exponential
// backoff. Keeping them apart is what lets the non-idempotent call sites reuse
// the loop under a stricter question (see `isSafeToReplay`).
//
// The classifier's default answer is no. For a backup tool, retrying a
// permanent failure spends round trips and delays every remaining file, while
// declining to retry a transient one costs a single file that the next run
// picks up anyway. When the evidence is ambiguous, fail fast.
import { ApiError, TruncatedStreamError } from "./errors.js";
export interface RetryOptions {
// Total calls, not retries: `attempts: 1` disables retrying.
attempts?: number;
// Ceiling for the first retry's delay; doubles with each retry.
baseDelayMs?: number;
// Upper bound on that ceiling, so a long outage settles into a steady
// poll instead of growing without limit.
maxDelayMs?: number;
// Injected so tests exercise the whole policy without waiting.
sleep?: (ms: number) => Promise<void>;
// Injected so the jitter is reproducible under test.
random?: () => number;
}
export type ResolvedRetryOptions = Required<RetryOptions>;
export const DEFAULT_RETRY_OPTIONS: ResolvedRetryOptions = {
attempts: 4,
baseDelayMs: 500,
maxDelayMs: 10_000,
sleep: (ms: number): Promise<void> =>
new Promise((resolve) => setTimeout(resolve, ms)),
random: Math.random,
};
export const resolveRetryOptions = (
opts?: RetryOptions,
): ResolvedRetryOptions => ({
attempts: opts?.attempts ?? DEFAULT_RETRY_OPTIONS.attempts,
baseDelayMs: opts?.baseDelayMs ?? DEFAULT_RETRY_OPTIONS.baseDelayMs,
maxDelayMs: opts?.maxDelayMs ?? DEFAULT_RETRY_OPTIONS.maxDelayMs,
sleep: opts?.sleep ?? DEFAULT_RETRY_OPTIONS.sleep,
random: opts?.random ?? DEFAULT_RETRY_OPTIONS.random,
});
// Transport failures: the request did not complete, for reasons below HTTP.
const TRANSPORT_CODES = new Set([
"ECONNRESET",
"ECONNABORTED",
"ETIMEDOUT",
"EPIPE",
"ENOTFOUND",
"EAI_AGAIN",
"ECONNREFUSED",
"EHOSTUNREACH",
"ENETUNREACH",
"ENETRESET",
"ENETDOWN",
]);
// The subset of the above that can only be reported before a TCP connection
// exists, and therefore before any request byte could have been written: name
// resolution produced no address (`ENOTFOUND`, `EAI_AGAIN`) or the peer
// refused the connection with an RST to the SYN (`ECONNREFUSED`).
//
// The routing errnos — `EHOSTUNREACH`, `ENETUNREACH`, `ENETDOWN` — are
// deliberately absent even though they look like connect-time failures. On
// Linux they are also delivered on an already-established socket: an ICMP
// destination-unreachable arriving mid-flight sets the socket error and the
// next read or write returns it, and a local interface going down after the
// request was fully written surfaces the same way. In those cases the server
// may already have received and acted on the request, which is exactly the
// ambiguity this set exists to exclude. They stay in `TRANSPORT_CODES`, so
// they remain retryable for idempotent calls; only replay eligibility is
// narrowed. See `isSafeToReplay`.
const CONNECT_CODES = new Set(["ENOTFOUND", "EAI_AGAIN", "ECONNREFUSED"]);
// `cause` is an arbitrary user-settable property and nothing prevents it from
// forming a cycle, so the walk is bounded. Hanging the process would be a
// worse outcome than any misclassification.
const MAX_CAUSE_DEPTH = 8;
// Collect every `code` in an error's cause chain. undici does not put the
// errno on the error it throws — it hangs the underlying socket error off
// `cause`, sometimes more than one level down — so a classifier that only read
// the top-level error would see a bare `Error` and call every dropped
// connection permanent. `complete` is false when the walk stopped at the
// depth limit with more of the chain still below it.
const causeCodes = (err: unknown): { codes: string[]; complete: boolean } => {
const codes: string[] = [];
let current: unknown = err;
for (let depth = 0; depth < MAX_CAUSE_DEPTH; depth++) {
if (current === null || typeof current !== "object") {
return { codes, complete: true };
}
const { code, cause } = current as { code?: unknown; cause?: unknown };
if (typeof code === "string") codes.push(code);
if (cause === current) return { codes, complete: true };
current = cause;
}
return { codes, complete: current === null || typeof current !== "object" };
};
const isAbort = (err: unknown): boolean => {
if (err === null || typeof err !== "object") return false;
const { name } = err as { name?: unknown };
return name === "AbortError" || name === "TimeoutError";
};
// Is another attempt capable of producing a different answer?
//
// A note on the truncation case, recorded on issue #2. `TruncatedStreamError`
// is retried, and for a body of more than one chunk that is exactly right: a
// chunk that failed to authenticate while the stream carried on past it is
// corruption, stays an ordinary authentication failure, and is not retried.
//
// For a single-chunk body — most thumbnails, every small file — the split is
// not achievable. A wrong file key, server-side corruption and a connection
// cut mid-chunk are cryptographically identical: Poly1305 fails and carries no
// framing signal. All three are reported as truncation and therefore retried.
// That imprecision is deliberate and bounded by the attempt count: one wasted
// round trip on a genuinely corrupt file is a fair price for never silently
// keeping a truncated one, and the alternative — treating a single-chunk
// authentication failure as permanent — would reintroduce exactly the
// silent-corruption risk the truncation check exists to remove.
export const isRetryable = (err: unknown): boolean => {
if (err instanceof ApiError) {
// 408 and 429 are the two 4xx codes that are statements about timing
// rather than about the request, and backoff is the right answer to
// both. Every other 4xx will answer the same way however often it is
// asked.
if (err.status === 408 || err.status === 429) return true;
return err.status >= 500 && err.status <= 599;
}
if (err instanceof TruncatedStreamError) return true;
// quak only ever aborts a request on its own deadline, so an abort means
// this attempt ran out of time — which a later one may not.
if (isAbort(err)) return true;
// Node's fetch rejects with `TypeError: fetch failed` for everything below
// HTTP: DNS failure, refused connection, TLS error, reset socket. Nothing
// on the object separates it from a TypeError thrown by a bug, so this is
// deliberately literal. Demanding a recognised `cause` instead would
// classify real network failures as permanent and fail backups that should
// have succeeded; the cost of the imprecision is bounded by the attempt
// count.
if (err instanceof TypeError) return true;
return causeCodes(err).codes.some((code) => TRANSPORT_CODES.has(code));
};
// Could the first attempt already have taken effect on the server?
//
// `isRetryable` is the wrong question for a request that changes state.
// `postJSON` and `putJSON` use this for every `POST` and `PUT` listed in the
// README under "Endpoints used"; verifying a second factor, for one, consumes
// one of a small number of attempts. They are replayed only on the failures in
// `CONNECT_CODES`, which establish that no TCP connection to the server ever
// existed: there was no address to connect to, or the peer refused the
// connection outright. A request byte cannot have been transmitted, so the
// server cannot have acted.
//
// Everything else is ambiguous. A 5xx proves the server did process the
// request. A reset or a broken pipe can arrive after it was fully sent and
// acted on. A routing errno can be delivered on an established socket. A
// deadline says nothing at all about the server's state. So every errno in the
// cause chain must be a connect errno: one other errno anywhere in the chain
// is doubt, and doubt is not replayed. A chain longer than the walk is doubt
// too: the links below the limit were never read.
export const isSafeToReplay = (err: unknown): boolean => {
const { codes, complete } = causeCodes(err);
return (
isRetryable(err) &&
complete &&
codes.length > 0 &&
codes.every((code) => CONNECT_CODES.has(code))
);
};
export interface WithRetryOptions extends RetryOptions {
isRetryable?: (err: unknown) => boolean;
}
// Exponential backoff with full jitter: the exponential term is the ceiling,
// and the actual wait is drawn uniformly below it. Full jitter, rather than a
// fixed delay plus noise, is what stops a client that lost a hundred parallel
// downloads to one CDN blip from re-sending all hundred at the same instant.
const backoffMs = (retryNumber: number, policy: ResolvedRetryOptions): number =>
policy.random() *
Math.min(policy.maxDelayMs, policy.baseDelayMs * 2 ** (retryNumber - 1));
export const withRetry = async <T>(
fn: () => Promise<T>,
opts?: WithRetryOptions,
): Promise<T> => {
const policy = resolveRetryOptions(opts);
const retryable = opts?.isRetryable ?? isRetryable;
for (let attempt = 1; ; attempt++) {
try {
return await fn();
} catch (err) {
if (attempt >= policy.attempts || !retryable(err)) {
// The error escapes unwrapped, and it is the one from the
// final attempt: callers classify what they catch, and
// `listMissingThumbnails` in particular needs the `ApiError`
// and its status intact.
throw err;
}
await policy.sleep(backoffMs(attempt, policy));
}
}
};