check / check (push) Successful in 14s
streamDecrypt no longer recopies the whole accumulation buffer on every network read. Reads are queued with a running byte count and a contiguous buffer is materialised only at each ENC_CHUNK_SIZE boundary, with a straddling read split via a subarray view, so each byte is copied once instead of O(n^2). TAG_FINAL truncation detection, final-chunk handling, retry, and the per-chunk progress hook are unchanged. A new test feeds a multi-chunk body through a ReadableStream that yields many small pieces, exercising the fragmented-read path. Model: opus-4-8
320 lines
13 KiB
TypeScript
320 lines
13 KiB
TypeScript
import { randomUUID } from "node:crypto";
|
|
import { open, rename, rm } from "node:fs/promises";
|
|
import type { FileHandle } from "node:fs/promises";
|
|
import { dirname, join } from "node:path";
|
|
import {
|
|
fromBase64,
|
|
initStreamPull,
|
|
pullStreamChunk,
|
|
STREAM_CHUNK_OVERHEAD,
|
|
STREAM_CHUNK_SIZE,
|
|
streamTagFinal,
|
|
} from "../crypto/index.js";
|
|
import { TruncatedStreamError } from "../errors.js";
|
|
import { withRetry } from "../retry.js";
|
|
import type { ApiClient } from "../api/client.js";
|
|
import type { EnteFile } from "../model/types.js";
|
|
|
|
export interface DownloadResult {
|
|
path: string;
|
|
bytesWritten: number;
|
|
}
|
|
|
|
// Fired as decrypted plaintext accumulates, with the running total of
|
|
// plaintext bytes recovered so far. Within one download it is non-decreasing
|
|
// and its last value equals the final `bytesWritten`. A retry restarts the
|
|
// file from byte zero (see `fetchAndDecrypt`), so a fresh attempt begins its
|
|
// own count from zero.
|
|
export type ProgressCallback = (bytesDone: number) => void;
|
|
|
|
const ENC_CHUNK_SIZE = STREAM_CHUNK_SIZE + STREAM_CHUNK_OVERHEAD;
|
|
|
|
// Decrypt a secretstream body, handing each plaintext chunk to `sink` as it is
|
|
// produced rather than accumulating the whole file. Peak memory is one
|
|
// ciphertext chunk of network buffer plus one plaintext chunk — bounded by
|
|
// `STREAM_CHUNK_SIZE` regardless of the file's size — so a multi-gigabyte video
|
|
// no longer needs its size again in RAM. Returns the total plaintext length.
|
|
//
|
|
// The truncation contract is exactly the buffered version's, only the sink is
|
|
// new: a body cut short still decrypts and authenticates up to its last whole
|
|
// chunk, so the absence of TAG_FINAL is the sole evidence it was cut short, and
|
|
// this throws rather than let a caller keep a short file. The sink has already
|
|
// seen those chunks by then; the caller (`decryptToTemp`) stages them in a temp
|
|
// file that is renamed into place only on a clean return, so a throw leaves
|
|
// nothing on disk.
|
|
const streamDecrypt = async (
|
|
stream: ReadableStream<Uint8Array>,
|
|
header: Uint8Array,
|
|
key: Uint8Array,
|
|
sink: (plaintext: Uint8Array) => Promise<void>,
|
|
onProgress?: ProgressCallback,
|
|
): Promise<number> => {
|
|
const state = initStreamPull(header, key);
|
|
const reader = stream.getReader();
|
|
// Incoming reads are held as-is and only stitched into a contiguous chunk
|
|
// at each `ENC_CHUNK_SIZE` boundary, so every received byte is copied once.
|
|
// Concatenating on each read instead — reallocating the whole accumulator
|
|
// per read — is O(n^2) in the bytes buffered, and for a 4 MiB chunk that
|
|
// memory churn dwarfs the libsodium decryption itself.
|
|
const pending: Uint8Array[] = [];
|
|
let pendingBytes = 0;
|
|
let totalPlain = 0;
|
|
let chunksPulled = 0;
|
|
let lastTag = -1;
|
|
|
|
// Remove the first `size` bytes from `pending` as one contiguous buffer.
|
|
// A read that straddles the boundary is split with `subarray` (a view, no
|
|
// copy); its tail stays queued for the next chunk. `size` never exceeds
|
|
// `pendingBytes`, so the queue always holds enough.
|
|
const takeContiguous = (size: number): Uint8Array => {
|
|
const out = new Uint8Array(size);
|
|
let offset = 0;
|
|
while (offset < size) {
|
|
const piece = pending[0]!;
|
|
const need = size - offset;
|
|
if (piece.length <= need) {
|
|
out.set(piece, offset);
|
|
offset += piece.length;
|
|
pending.shift();
|
|
} else {
|
|
out.set(piece.subarray(0, need), offset);
|
|
pending[0] = piece.subarray(need);
|
|
offset += need;
|
|
}
|
|
}
|
|
pendingBytes -= size;
|
|
return out;
|
|
};
|
|
|
|
const consume = async (
|
|
plaintext: Uint8Array,
|
|
tag: number,
|
|
): Promise<void> => {
|
|
await sink(plaintext);
|
|
totalPlain += plaintext.length;
|
|
chunksPulled++;
|
|
lastTag = tag;
|
|
onProgress?.(totalPlain);
|
|
};
|
|
|
|
for (;;) {
|
|
const { done, value } = await reader.read();
|
|
if (value && value.length > 0) {
|
|
pending.push(value);
|
|
pendingBytes += value.length;
|
|
}
|
|
|
|
while (pendingBytes >= ENC_CHUNK_SIZE) {
|
|
const encChunk = takeContiguous(ENC_CHUNK_SIZE);
|
|
// A whole chunk that fails to authenticate while the stream carries
|
|
// on is corruption, not truncation; that error propagates unchanged.
|
|
const { plaintext, tag } = pullStreamChunk(state, encChunk);
|
|
await consume(plaintext, tag);
|
|
}
|
|
|
|
if (done) {
|
|
if (pendingBytes > 0) {
|
|
const buffer = takeContiguous(pendingBytes);
|
|
// Whatever is left over once every whole chunk has been
|
|
// consumed must be the stream's final chunk, and a final
|
|
// chunk that actually arrived in full authenticates. If it
|
|
// does not, the body stopped part-way through a chunk — the
|
|
// ordinary shape of a dropped connection. Poly1305 cannot
|
|
// tell a partial chunk from a corrupt one, so this is
|
|
// reported as the truncation it almost always is, with the
|
|
// authentication failure kept as the error's cause. Only the
|
|
// pull is guarded: a sink failure on a chunk that did
|
|
// authenticate is a disk error, not a truncation.
|
|
let pulled;
|
|
try {
|
|
pulled = pullStreamChunk(state, buffer);
|
|
} catch (err) {
|
|
throw new TruncatedStreamError(
|
|
`download: stream truncated: response body ended with ${buffer.length} trailing bytes that did not authenticate as a final chunk (transfer stopped mid-chunk, or the data is corrupt)`,
|
|
{ cause: err },
|
|
);
|
|
}
|
|
await consume(pulled.plaintext, pulled.tag);
|
|
}
|
|
break;
|
|
}
|
|
}
|
|
|
|
// Only the last chunk of a secretstream carries TAG_FINAL. Everything a
|
|
// dropped connection did deliver still decrypts and authenticates, so the
|
|
// absence of TAG_FINAL is the only evidence that the body was cut short.
|
|
if (chunksPulled === 0) {
|
|
throw new TruncatedStreamError(
|
|
"download: stream truncated: response body contained no secretstream chunks",
|
|
);
|
|
}
|
|
const tagFinal = streamTagFinal();
|
|
if (lastTag !== tagFinal) {
|
|
throw new TruncatedStreamError(
|
|
`download: stream truncated: last chunk tag ${lastTag}, expected TAG_FINAL (${tagFinal})`,
|
|
);
|
|
}
|
|
return totalPlain;
|
|
};
|
|
|
|
// Stage a write to `destination` atomically and durably, then rename it into
|
|
// place. `fill` writes the contents into the open temp file handle — either the
|
|
// whole buffer at once (`writeAtomic`) or chunk by chunk as they decrypt
|
|
// (`decryptToTemp`). The temp file is a sibling of the destination (same
|
|
// directory, so the rename cannot cross a filesystem boundary), so callers
|
|
// never observe a partially written destination, and a pre-existing file is
|
|
// replaced only once the new contents are complete on disk.
|
|
//
|
|
// Durability against a power cut needs two fsyncs. Without them the write can
|
|
// return while the data or the rename is still only in the kernel's page
|
|
// cache, and a crash then resurrects an empty renamed file — exactly the
|
|
// corruption a later backup run treats as a complete download. So the temp
|
|
// file's contents are fsynced before the rename, and the containing directory
|
|
// is fsynced after it, so both the bytes and the new directory entry are on
|
|
// stable storage before this returns.
|
|
//
|
|
// On any failure — including a `fill` that throws because the stream was
|
|
// truncated — the temp file is removed, so the destination is untouched and no
|
|
// scratch file is left to fill the disk on repeated failures.
|
|
const stageAtomic = async (
|
|
destination: string,
|
|
fill: (handle: FileHandle) => Promise<void>,
|
|
): Promise<void> => {
|
|
const dir = dirname(destination);
|
|
// The random suffix keeps concurrent downloads of the same destination
|
|
// from stepping on each other's temporary file.
|
|
const tmpPath = join(dir, `.quak-${randomUUID()}.tmp`);
|
|
try {
|
|
const handle = await open(tmpPath, "w");
|
|
try {
|
|
await fill(handle);
|
|
await handle.sync();
|
|
} finally {
|
|
await handle.close();
|
|
}
|
|
await rename(tmpPath, destination);
|
|
// Fsync the directory so the rename itself survives a crash: renaming
|
|
// over a synced temp file still leaves the new directory entry in the
|
|
// page cache until the directory is synced.
|
|
const dirHandle = await open(dir, "r");
|
|
try {
|
|
await dirHandle.sync();
|
|
} finally {
|
|
await dirHandle.close();
|
|
}
|
|
} catch (err) {
|
|
// Best-effort cleanup. A failure to remove the temporary file must
|
|
// never replace the error that actually explains what went wrong.
|
|
await rm(tmpPath, { force: true }).catch(() => undefined);
|
|
throw err;
|
|
}
|
|
};
|
|
|
|
// Write `plaintext` to `destination` atomically and durably. Exported so the
|
|
// metadata store can reuse the same durable write for small whole-buffer
|
|
// payloads; originals go through `decryptToTemp` instead so they never buffer.
|
|
export const writeAtomic = async (
|
|
destination: string,
|
|
plaintext: Uint8Array,
|
|
): Promise<void> =>
|
|
stageAtomic(destination, (handle) => handle.writeFile(plaintext));
|
|
|
|
// Decrypt `stream` straight to `destination`, one plaintext chunk at a time,
|
|
// under the atomic writer's temp-then-rename discipline. Memory stays bounded
|
|
// by the chunk size: each decrypted chunk is written to the temp file and
|
|
// dropped. The rename happens only after the stream authenticates as terminated
|
|
// on TAG_FINAL; a truncated stream throws and leaves the destination untouched.
|
|
// Returns the plaintext length written.
|
|
const decryptToTemp = async (
|
|
destination: string,
|
|
stream: ReadableStream<Uint8Array>,
|
|
header: Uint8Array,
|
|
key: Uint8Array,
|
|
onProgress?: ProgressCallback,
|
|
): Promise<number> => {
|
|
let bytesWritten = 0;
|
|
await stageAtomic(destination, async (handle) => {
|
|
bytesWritten = await streamDecrypt(
|
|
stream,
|
|
header,
|
|
key,
|
|
async (plaintext) => {
|
|
await handle.write(plaintext);
|
|
},
|
|
onProgress,
|
|
);
|
|
});
|
|
return bytesWritten;
|
|
};
|
|
|
|
// Fetch a stream and decrypt it to `destination`, retrying the whole sequence.
|
|
//
|
|
// The request is only the first third of a download. `getXStream` returns as
|
|
// soon as headers arrive, and the bytes are pulled here, so a socket reset
|
|
// mid-body — the dominant failure mode for multi-megabyte photos over a CDN —
|
|
// throws in `streamDecrypt` and never reaches `ApiClient` at all. Retrying the
|
|
// request alone would miss it entirely.
|
|
//
|
|
// The client's own retry is therefore switched off for these two calls: with
|
|
// both layers active the budgets would multiply, and the library default of
|
|
// four attempts would mean sixteen requests for one file. The policy comes
|
|
// from the client so a caller that configured one gets it here too.
|
|
//
|
|
// Because the plaintext is streamed to disk rather than buffered, the atomic
|
|
// write is part of the retried unit. A retry starts the file over from byte
|
|
// zero — the secretstream pull state is not resumable and there is no Range
|
|
// support — staging into a fresh temp file each time: a failed attempt writes
|
|
// and then removes its own temp file, and only the attempt that reaches
|
|
// TAG_FINAL renames one into place, so a download that needed three tries still
|
|
// performs exactly one rename over the destination.
|
|
const fetchAndDecrypt = async (
|
|
api: ApiClient,
|
|
openStream: () => Promise<ReadableStream<Uint8Array>>,
|
|
header: Uint8Array,
|
|
key: Uint8Array,
|
|
destination: string,
|
|
onProgress?: ProgressCallback,
|
|
): Promise<number> =>
|
|
withRetry(async () => {
|
|
const stream = await openStream();
|
|
return decryptToTemp(destination, stream, header, key, onProgress);
|
|
}, api.getRetryOptions());
|
|
|
|
export const downloadFile = async (
|
|
api: ApiClient,
|
|
file: EnteFile,
|
|
outPath?: string,
|
|
onProgress?: ProgressCallback,
|
|
): Promise<DownloadResult> => {
|
|
const resolvedPath = outPath ?? file.metadata.title;
|
|
const header = fromBase64(file.file.decryptionHeader);
|
|
const bytesWritten = await fetchAndDecrypt(
|
|
api,
|
|
() => api.getFileStream(file.id, { retry: false }),
|
|
header,
|
|
file.key,
|
|
resolvedPath,
|
|
onProgress,
|
|
);
|
|
return { path: resolvedPath, bytesWritten };
|
|
};
|
|
|
|
export const downloadThumbnail = async (
|
|
api: ApiClient,
|
|
file: EnteFile,
|
|
outPath?: string,
|
|
onProgress?: ProgressCallback,
|
|
): Promise<DownloadResult> => {
|
|
const resolvedPath = outPath ?? `thumb_${file.metadata.title}`;
|
|
const header = fromBase64(file.thumbnail.decryptionHeader);
|
|
const bytesWritten = await fetchAndDecrypt(
|
|
api,
|
|
() => api.getThumbnailStream(file.id, { retry: false }),
|
|
header,
|
|
file.key,
|
|
resolvedPath,
|
|
onProgress,
|
|
);
|
|
return { path: resolvedPath, bytesWritten };
|
|
};
|