1 Commits
Author SHA1 Message Date
sneak e7f77b4b27 Bound the originals cache with adaptive LRU eviction (closes #47)
check / check (push) Successful in 35s
Bounds cacheDirectory/originals to an adaptive limit: min(configured max,
bytesUsed + bytesFree - reserve), bytesFree from statfs, so it tracks disk
pressure (status().originalsLimitBytes exposes it). Each original write evicts
least-recently-used originals until usage fits; last-use is the file mtime,
bumped on every read that returns a path, so order survives restarts.

Never evicted: pinned originals and every original whose write overlaps the
evicting one — its own and any concurrent sibling's, since content writes run in
parallel. So an over-budget fetch keeps the path it returns, and no write can
delete a sibling's file before that fetch returns it; such files become eligible
on a later, non-overlapping write. Defaults: 100 GiB limit, 50 GiB reserve.

Model: opus-4-8
2026-09-22 17:57:20 +00:00
72 changed files with 2550 additions and 9081 deletions
+1 -1
View File
@@ -1,6 +1,6 @@
# Mirrors .gitignore, with one deliberate exception: .gitignore itself stays
# in the build context, because prettier 3 reads it as a default ignore file
# and dropping it would change what the lint phase's prettier check sees.
# and dropping it would change what `make fmt-check` sees inside the image.
# VCS
.git
+17 -58
View File
@@ -1,69 +1,28 @@
# Lint phase. The linters are invoked directly rather than through `make
# lint` or `script/lint`, which are themselves a docker build and would
# recurse into a daemon that does not exist in a build step.
# Test and build image: the suite, then the compile.
#
# Linting deliberately does not happen here. `script/lint` is a build of
# Dockerfile.lint, and `script/check` calls `script/lint`, so running
# `make check` in this image would mean running `docker build` inside a
# container. Lint runs exactly once, in Dockerfile.lint; script/cibuild
# builds that first and this second.
# node 22.22.0 on Alpine 3.23.3 (node:22-alpine), 2026-08-09
FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS lint
FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS check
WORKDIR /app
COPY script/ script/
COPY package.json yarn.lock ./
RUN script/bootstrap
COPY . .
RUN yarn run eslint .
RUN yarn run prettier --check .
# Test phase, same shape and for the same reason. The suite runs without
# verbose output first and is rerun verbosely only if it fails; the timeout
# catches a hung test.
#
# node 22.22.0 on Alpine 3.23.3 (node:22-alpine), 2026-08-09
FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS test
WORKDIR /app
COPY script/ script/
COPY package.json yarn.lock ./
RUN script/bootstrap
COPY . .
# Unlike the template, the suite runs as the image's non-root `node` user:
# root ignores directory permissions, so the tests of a destination that is
# not writable would otherwise fail. vitest writes into /app.
RUN chown -R node:node /app
USER node
RUN timeout 90 yarn run vitest run --reporter=dot || \
{ echo "--- Rerunning with verbose for details ---"; \
timeout 90 yarn run vitest run --reporter=verbose; exit 1; }
# Build stage, and the last stage: a plain `docker build .` names no target
# and so builds this one. Nothing is wanted from the two phases above; the
# copies are what make BuildKit build them first, so this image cannot be
# produced unless lint and test passed. A stage appended after this one
# would drop all three out of a plain build.
#
# node 22.22.0 on Alpine 3.23.3 (node:22-alpine), 2026-08-09
FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34
WORKDIR /app
COPY --from=lint /app/package.json /dev/null
COPY --from=test /app/package.json /dev/null
COPY script/ script/
COPY package.json yarn.lock ./
RUN script/bootstrap
COPY . .
# The version is computed on the host and passed in, because
# .dockerignore excludes .git.
ARG VERSION=dev
LABEL org.opencontainers.image.version="${VERSION}"
# CHECK_EPOCH is a cache buster: without it Docker serves the test layer from
# cache on an unchanged tree, the suite never executes, and the build still
# exits 0. The guard makes an absent argument a hard failure — an unset ARG
# is the empty string, which is a perfectly stable cache key, so a plain
# `docker build .` would otherwise still get the false green. Fail closed.
ARG CHECK_EPOCH
RUN [ -n "$CHECK_EPOCH" ] || exit 1
RUN make test
ARG CHECK_EPOCH
RUN [ -n "$CHECK_EPOCH" ] || exit 1
RUN make build
+35
View File
@@ -0,0 +1,35 @@
# Lint image: every lint run happens here, and nowhere else. The repo is
# COPYed into a digest-pinned image and the linters run as build steps, so a
# successful build IS a clean lint. `script/lint` does nothing but build this
# file, which also works where the docker daemon is remote and bind mounts are
# impossible. Nothing that runs inside a container may call `script/lint`:
# that is why Dockerfile no longer runs `make check`.
# node 22.22.0 on Alpine 3.23.3 (node:22-alpine), 2026-08-09
FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS lint
WORKDIR /app
# Manifests before sources, so the dependency install layer stays cached
# until package.json or yarn.lock changes. script/bootstrap ends in
# `yarn install --frozen-lockfile`; the lint steps below are deliberately
# not cached.
COPY script/ script/
COPY package.json yarn.lock ./
RUN script/bootstrap
COPY . .
# LINT_EPOCH is a cache buster, with the same fail-closed contract as
# CHECK_EPOCH in Dockerfile. No lint cache is wanted: on an unchanged tree
# Docker serves the linter layers in well under a second, having linted
# nothing, and the build still exits 0. The guard makes an absent argument a
# hard failure — an unset ARG is the empty string, which is a perfectly
# stable cache key, so a plain `docker build -f Dockerfile.lint .` would
# otherwise get exactly that false green. Every layer below this one is a
# child of the guard, so a fresh epoch forces all of them to execute.
ARG LINT_EPOCH
RUN [ -n "$LINT_EPOCH" ] || exit 1
# The linters are invoked directly rather than through `make lint`, because
# `make lint` is the build of this file.
RUN yarn run eslint .
RUN yarn run prettier --check .
+130 -399
View File
@@ -38,47 +38,34 @@ yarn quak get 67890 --out ./photo.jpg
yarn quak backup ./my-backup
```
For library use, the primary surface is the cache-backed `Library`:
For library use:
```ts
import { Client, Library } from "quak";
import { Client } from "quak";
// Log in once; the client satisfies the library's client interface.
const client = await Client.login({
email: "you@example.com",
password: "your-password",
});
// Open a cache-backed library. On an empty cache this awaits one server
// refresh; on an existing cache it returns immediately and refreshes in the
// background every `refreshIntervalSeconds` (default 3).
const lib = await Library.open({ client });
// Default reads answer synchronously from the local cache — no network.
for (const album of lib.albums.list()) {
console.log(album.collectionID, album.name);
for (const photo of album.photos.list()) {
console.log(` ${photo.title} [${photo.fileType}]`);
for (const c of await client.listCollections()) {
console.log(c.id, c.name);
const files = await client.listFiles(c.id, c.key);
for (const f of files) {
console.log(` ${f.metadata.title} [${f.metadata.fileType}]`);
}
}
// Fresh reads await a server round-trip and answer with current state.
const { albums } = await lib.fresh();
console.log(`${albums.list().length} albums as of now`);
// Download a file
const files = await client.listFiles(collectionID, collectionKey);
await client.downloadFile(files[0], "./photo.jpg");
// Fetch (and cache) one photo's full-resolution bytes.
const photo = lib.photos.byID({ fileID: 12345 });
if (photo) {
const { path } = await photo.original();
console.log(`original at ${path}`);
}
await lib.close();
// Serialize session for later (consumer handles persistence)
const snapshot = client.toJSON();
// ... later:
const restored = Client.fromJSON(snapshot);
```
The lower-level `Client` (login, session serialization, and the raw
enumeration/download calls) is exported too and documented under Design below.
## Entrypoints
This repository adheres to the
@@ -97,18 +84,20 @@ alpine. We provide:
- `script/build` — compile the TypeScript sources into `dist/`, then verify that
the entrypoints `package.json` declares (`main`, `types`, `bin`) are among the
files the compiler wrote, and make the CLI executable (our own extension)
- `script/test` — run the test suite, by building the `test` phase of the
`Dockerfile` (vitest, 90s timeout, verbose rerun on failure); requires docker
- `script/lint` — run eslint and a prettier check, by building the `lint` phase
of the `Dockerfile`; requires docker (see Linting and testing below)
- `script/test` — run the test suite (vitest, hard-capped at 30s where `timeout`
is available, verbose rerun on failure)
- `script/lint` — run eslint and a prettier check, by building
`Dockerfile.lint`; requires docker (see Linting below)
- `script/fmt` — format all files with prettier (writes)
- `script/fmt-check` — check formatting on the host (read-only); standalone, and
not called by `script/check` or `script/precommit`, because `script/lint`
already checks formatting in the container
already checks formatting in the container (see Linting below)
- `script/check` — run all checks: `test`, `lint` (our own extension)
- `script/docker` — build the image, tagged via `script/projectname`
- `script/cibuild` — build the image (what CI runs); its last stage depends on
the `lint` and `test` phases, so this one build lints, tests and compiles
- `script/docker` — build the test and build image, tagged via
`script/projectname`
- `script/cibuild` — cd to the repo root and build both images (what CI runs):
`script/lint` first, then the `Dockerfile` image, which runs `make test` and
`make build`
- `script/precommit` — run by the git pre-commit hook (our own extension); runs
`script/lint`, which checks both lint and formatting, but deliberately not the
tests, so the TDD red-phase commit can land
@@ -117,32 +106,50 @@ alpine. We provide:
`make hooks` installs the pre-commit hook that runs `script/precommit`.
### Linting and testing
### Linting
Linting and testing are phases of the `Dockerfile`. The `lint` phase copies the
repo into a digest-pinned node image and runs eslint and `prettier --check .`;
the `test` phase does the same with the suite. `script/lint` and `script/test`
each build one phase with `docker build --no-cache --target <phase>`. There is
no host lint or test path: docker is required, and that also works where the
docker daemon is remote and bind mounts are impossible.
Linting runs in a container, one way, everywhere. `script/lint` builds
`Dockerfile.lint`, which copies the repo into a digest-pinned node image and
runs eslint and prettier as build steps, so a successful build is a clean lint.
There is no host lint path: docker is required to lint, and that also works
where the docker daemon is remote and bind mounts are impossible.
The last stage of the `Dockerfile` compiles the package, and it copies a file
from each phase, so it cannot be built unless lint and the tests pass. That is
why `script/cibuild` is a single `docker build`: it runs lint and the tests once
each and then compiles.
The formatting check is part of that, not a step beside it. `script/check` and
`script/precommit` therefore call `script/lint` and stop; neither calls
`script/fmt-check` as well, which would run prettier a second time over the same
tree for the same verdict — and the weaker of the two, since the host's prettier
is whatever the working tree has installed. So `make check` and the pre-commit
hook both still fail on a badly formatted tree, and prettier runs exactly once
in each. `test/packaging/lint-once.test.ts` asserts that count by walking the
invocation graph, so a second pass cannot creep back in unnoticed.
Every `docker build` in `script/` passes `--no-cache`. On an unchanged tree
Docker would otherwise serve the lint and test steps from cache, nothing would
run, and the build would still exit 0.
The formatting check is part of the `lint` phase, not a step beside it, so
`script/check` and `script/precommit` do not call `script/fmt-check` as well;
that would run prettier a second time over the same tree for the same verdict.
`script/fmt-check` remains as a standalone entrypoint for asking the formatting
question on the host. Its verdict matches the container's: prettier is pinned to
an exact version, installed from `yarn.lock` under `--frozen-lockfile` in both
places, and reads `.gitignore` as its default ignore file — which is why
`.dockerignore` keeps `.gitignore` in the build context.
question on its own, without docker and without the rest of lint. Its verdict
cannot drift from the container's: prettier is pinned to an exact version,
installed from `yarn.lock` under `--frozen-lockfile` in both places, and reads
`.gitignore` as its default ignore file — which is why `.dockerignore`
deliberately keeps `.gitignore` in the build context.
Lint happens in exactly one place, which constrains the rest of the build.
`script/check` calls `script/lint`, so `make check` cannot run inside a
container without asking for docker inside docker. The image built from
`Dockerfile` therefore runs `make test` and `make build` and does not lint;
`script/cibuild` builds `Dockerfile.lint` first and that image second, so CI
gets both verdicts.
### Build epochs
`script/lint` passes `--build-arg LINT_EPOCH="$(date +%s)"`, and `script/docker`
and `script/cibuild` pass `--build-arg CHECK_EPOCH="$(date +%s)"`. Both
Dockerfiles refuse to build without their argument. This is deliberate: on an
unchanged tree Docker would otherwise serve the linter and test layers from
cache, so nothing would run and the build would still exit 0 — a lint build over
an untouched tree returns success in well under a second, having linted nothing.
A changing epoch invalidates every layer below the guard on every invocation
while leaving the dependency layers above them cached, and the missing-argument
guard means a bare `docker build .` fails loudly instead of quietly reporting a
green it did not earn: an unset build argument is the empty string, which is a
perfectly stable cache key.
## Rationale
@@ -177,9 +184,10 @@ All work on quak is test-driven. No exceptions.
3. Subsequent commits add the implementation and any refactors needed to make
the tests pass.
4. A feature branch can only be merged into `main` when `make check` is green.
`main` is always green. CI runs `script/cibuild`, which builds the
`Dockerfile`: its `lint` and `test` phases, then the compile, so neither a
red branch nor one that does not compile can pass CI.
`main` is always green. CI runs `script/cibuild`, which lints via
`Dockerfile.lint` and then runs `make test` and `make build` in the
`Dockerfile` image, so neither a red branch nor one that does not compile can
pass CI.
5. Tests are the canonical API documentation for this library. Every test file
is commented thoroughly enough that a reader who has never seen quak can
learn how to use it from the tests alone. Comments explain why a behavior
@@ -196,7 +204,7 @@ All work on quak is test-driven. No exceptions.
runs `script/lint` — eslint and the prettier check, in the container — but
not the tests, and so not the full `make check`. This is deliberate so the
TDD red-phase commit (failing tests, no implementation yet) can land. The
`test` phase is part of the image build, which is what CI executes via
suite runs as part of the image build, which is what CI executes via
`script/cibuild`, so a red branch still cannot reach `main`.
## Design
@@ -214,29 +222,18 @@ quak/
auth/ login flow (SRP + email OTP + TOTP), key unwrap
model/ decrypted Collection, File, Metadata types + decrypt fns
download/ streaming file/thumbnail download + decryption
library/ the cache-backed Library: metadata store, read
surface, records, content cache, precache, ML data
and search, request pools
backup.ts resilient full-account backup with dedup
metadata-backup.ts
backup-metadata: all decrypted metadata as JSON
mldata-fetch.ts fetch + decrypt per-file ML data
filename.ts safe file names from server metadata
errors.ts error types shared across layers
retry.ts retry classifier + exponential backoff with jitter
thumbnails.ts detect + regenerate missing thumbnails
client.ts high-level Client class assembled from the above
cli-commands.ts the CLI's commands as functions returning exit codes
cli-output.ts how the CLI prints a file's title and time
cli-read.ts fresh reads for the CLI's read commands
cli-run.ts run a command, print its error, exit with its code
cli-session.ts read the saved session file back into a Client
index.ts public library exports
bin/
quak.ts CLI entrypoint (commander.js)
test/ unit + integration tests (vitest)
Makefile
Dockerfile lint phase, test phase, compile
Dockerfile test suite and compile
Dockerfile.lint eslint and prettier, as build steps
package.json
tsconfig.json
```
@@ -308,13 +305,11 @@ Endpoints used:
encrypted token plus key attributes.
- `POST /users/ott` and `POST /users/verify-email`: email OTP fallback path.
- `POST /users/two-factor/verify`: TOTP second factor.
- `POST /users/logout`: end the calling token's session (`quak logout`).
- `GET /collections/v2?sinceTime=<usec>`: list collections changed since
microsecond timestamp; pass 0 for a full enumeration.
- `GET /collections/v2/diff?collectionID=<id>&sinceTime=<usec>`: list files in a
collection; paginate while `hasMore` is true.
- `GET https://files.ente.io/?fileID=<id>`: download encrypted file bytes.
- `POST /files/data/fetch`: fetch encrypted ML data for a batch of files.
- `POST /files/upload-url`: mint a presigned upload URL (for thumbnail repair).
- `PUT /files/thumbnail`: register an uploaded thumbnail's object key.
@@ -354,41 +349,35 @@ and a half seconds of waiting. `sleep` and `random` are injectable through the
same option, which is how the test suite exercises the whole policy without
waiting.
Two deadlines, renewed for each attempt:
Two deadlines, applied with `AbortSignal.timeout()` and renewed for each
attempt:
| Option | Default | Applies to | Kind |
| ------------------- | ------- | ------------------------------------------- | ------------------------------------- |
| `requestTimeoutMs` | `30000` | `getJSON`, `postJSON`, `putJSON`, `putFile` | the whole request |
| `downloadTimeoutMs` | `60000` | file and thumbnail downloads | idle: no bytes received for this long |
| Option | Default | Applies to |
| ------------------- | -------- | ------------------------------------------- |
| `requestTimeoutMs` | `30000` | `getJSON`, `postJSON`, `putJSON`, `putFile` |
| `downloadTimeoutMs` | `600000` | file and thumbnail body transfers |
They are different kinds because a download's length depends on the file and the
link: a whole-transfer deadline short enough to catch a hung connection would
cancel a large video on a slow link that is still making progress. The download
deadline restarts every time bytes arrive, so a slow download runs as long as it
keeps moving, and one that stalls is aborted after 60 seconds of silence. It
covers the wait for the headers and the body — `getFileStream` returns as soon
as headers arrive, so a deadline that only guarded the initial request would
leave the same hang one layer down. There is no limit on the total length of a
download.
They are separate because one number cannot serve both: a value short enough to
keep a hung API call from stalling a backup would cancel a legitimate
multi-gigabyte download. The download deadline covers the body, not just the
headers — `getFileStream` returns as soon as headers arrive, so a deadline that
only guarded the initial request would leave the same hang one layer down.
**Non-idempotent requests are not blindly replayed.** `postJSON` and `putJSON`
send every `POST` and `PUT` in the endpoint list above; some of them change
server state, and `/users/two-factor/verify` consumes one of a small number of
second-factor attempts. They are retried only when every errno in the error's
`cause` chain is one of the three that establish no TCP connection to the server
ever existed, so no request byte can have been transmitted: `ENOTFOUND` and
`EAI_AGAIN` (name resolution produced no address) and `ECONNREFUSED` (the peer
refused the connection). A 5xx, a mid-flight reset and a deadline are all left
to the caller, because each of them can happen after the server has already
acted. These two do not follow redirects either: a redirect means the server
already received the request, so it is reported as an error and not retried. The
routing errnos `EHOSTUNREACH`, `ENETUNREACH` and `ENETDOWN` are excluded for the
same reason, despite looking like connect-time failures: on Linux an ICMP
unreachable arriving mid-flight, or a local interface going down after the
request was written, delivers them on an already-established socket. They stay
retryable for the idempotent calls. `putFile` is exempt: a presigned PUT stores
one whole object at one key in one request, so replaying it has no partial state
to damage.
reach `/users/srp/create-session`, `/users/two-factor/verify` — which consumes
one of a small number of second-factor attempts — and `/files/thumbnail`. They
are retried only on the three failures that establish no TCP connection to the
server ever existed, so no request byte can have been transmitted: `ENOTFOUND`
and `EAI_AGAIN` (name resolution produced no address) and `ECONNREFUSED` (the
peer refused the connection). A 5xx, a mid-flight reset and a deadline are all
left to the caller, because each of them can happen after the server has already
acted. The routing errnos `EHOSTUNREACH`, `ENETUNREACH` and `ENETDOWN` are
excluded for the same reason, despite looking like connect-time failures: on
Linux an ICMP unreachable arriving mid-flight, or a local interface going down
after the request was written, delivers them on an already-established socket.
They stay retryable for the idempotent calls. `putFile` is exempt: a presigned
PUT stores one whole object at one key in one request, so replaying it has no
partial state to damage.
A download is retried as a whole — request, stream consumption, and decryption —
because a socket reset after the response headers have arrived surfaces in the
@@ -420,77 +409,32 @@ decides how to persist sessions.
`client.toJSON()` returns a `ClientSnapshot` (a plain serializable object with
base64-encoded keys) that the consumer can write to disk, a database, or
whatever else fits their use case. `Client.fromJSON(snapshot)` restores a
working client from that snapshot without re-authenticating; it checks every
field and each key's length first, and throws an error naming the bad field.
`client.logout()` clears the token and zeroes the key buffers in place; every
later call on that client throws. It does not contact the server, so the token
stays valid there and in any saved snapshot; `await client.logoutOnServer()`
first ends the session on the server (`POST /users/logout`).
working client from that snapshot without re-authenticating.
The CLI stores the snapshot at the platform-appropriate data directory via
`env-paths`: `~/Library/Application Support/quak/session.json` on macOS,
`$XDG_DATA_HOME/quak/session.json` on Linux. The file is written with mode
`0600`. The key material is stored in cleartext in the JSON; treat this file as
you would treat the password itself. A missing file is reported as "not logged
in"; a file that exists but is corrupt is reported as such, naming the bad
field. Both exit with status 1.
`quak logout` ends the session on the server, so the token in `session.json`
stops working even in a copy of the file, and then deletes the file. If the
server call fails (or the file is corrupt), the file is still deleted, the
command says the server session could not be ended, and it exits with status 1.
It does not delete the cache: it prints the account's cache directory and says
it still holds decrypted data (file keys in `metadata.json`, cached originals
and thumbnails), for the user to delete if they want it gone.
you would treat the password itself.
### CLI surface
```
quak [--cache-dir <path>] <command> global: local metadata/content cache location
quak login interactive or QUAK_EMAIL/QUAK_PASSWORD
quak whoami print logged-in account as JSON
quak logout end the session, delete it
quak logout delete saved session
quak collections [--json] list all collections
quak files --collection <id> [--json] list files in a collection
quak get <fileID> [--out path] [--collection] download and decrypt a file
quak get-thumb <fileID> [--out] [--collection] download and decrypt a thumbnail
quak backup <dir> [--json] full incremental backup
quak backup-metadata <dir> [--exif] dump all decrypted metadata as JSON
quak helper list-missing-thumbnails [--json] find files with missing thumbnails
quak helper fix-missing-thumbnails [--file ids] generate + upload missing thumbnails
```
Every command runs on the same cache-backed library. The read commands —
`collections`, `files`, `get`, `get-thumb`, `backup-metadata`,
`helper list-missing-thumbnails` and `helper fix-missing-thumbnails` — force a
fresh server round-trip before they answer, so they report current account state
rather than whatever the cache last held. If that round-trip fails, the command
prints the error on one line and exits 1. `--cache-dir` overrides where the
cache lives; without it each account gets its own directory under the per-user
cache path.
`get` and `get-thumb` resolve the file by ID directly, so `--collection` is
accepted for backward compatibility but ignored. `backup-metadata --exif` (alias
`--all`) additionally downloads each file to extract full EXIF/IPTC/XMP
metadata. The listing and backup commands support `--json` for machine-readable
output.
`backup-metadata` fetches ML data in requests of up to 200 files. When a request
still fails after its retries, the error is logged, each of its files is written
with the reason in an `mlDataError` field instead of `mlData`, and the dump goes
on. The exit code is non-zero if any ML data request failed.
`helper fix-missing-thumbnails` regenerates thumbnails for baseline JPEG images
only, because the bundled decoder (`jpeg-js`) decodes only JPEG. A non-JPEG
image (PNG, HEIC) or a video is reported as `skipped` (unsupported format), kept
distinct from a `failed` repair, and does not affect the exit code; a genuine
failure still exits non-zero. The server accepts a new thumbnail only from the
file's owner and only when it is no larger than the thumbnail size it records
for the file. So a file another account owns, in an album shared with you, is
skipped by both thumbnail helpers without being fetched, and the fixer skips a
file whose recorded thumbnail size is 0 or unknown. Otherwise the fixer lowers
the quality and size of the thumbnail until it fits, and skips the file if even
the smallest does not.
`get` and `get-thumb` search all collections for the file ID when `--collection`
is not specified. All listing and backup commands support `--json` for
machine-readable output.
### Backup layout
@@ -505,62 +449,20 @@ the smallest does not.
<name>/
<title> -> ../../originals/<fileID>.<ext> (symlink)
<name>.json collection metadata + file list
failures.json files that failed and have not yet succeeded
```
`failures.json` records each failed file with the kind of failure, how many
times it has been tried and when it was last tried. A file leaves it once it
succeeds, or once it is no longer in the library or in the backup's scope. The
library's `lib.backup({ includeThumbnails: true })` also writes
`thumbnails/<fileID>.jpg` beside `originals/`; `quak backup` does not.
A collection's directory and JSON are named after the collection, and a symlink
after the file's title, both with unsafe characters replaced. When two
collections would get the same name, or two files in one collection the same
title (ignoring case in both), each of them gets its ID added: two albums named
`Trip` become `Trip (10)/` and `Trip (11)/`, and two files titled `IMG_0001.JPG`
become `IMG_0001 (12345).JPG` and `IMG_0001 (12346).JPG`. IDs never change, so a
name stays the same from run to run until such a clash appears or goes away.
Each run removes the symlinks into `originals/` that no longer belong in their
collection's directory, and the directories (and JSON) of collections that were
deleted or renamed. Nothing else in `collections/` is touched: a file or a
symlink you put there stays, and a directory that still holds one after its
symlinks are removed stays too, with its JSON.
Each file is downloaded exactly once regardless of how many collections it
appears in, and written once: straight into `originals/`, with no copy left in
the cache. An original the cache already held is copied from there instead. On
subsequent runs, existing originals are skipped. If a download fails, the error
is logged and the backup continues with the next file. The exit code is non-zero
if any files failed. `quak backup` opens its library with the thumbnail and
originals precache off, so it fetches only what the backup stores.
Each original is written to a temporary file in the same directory, synced to
disk, and renamed into place, so an original is either complete or absent, even
after a power cut. A downloaded original's temporary file is named
`.quak-<pid>-<random>.tmp`, one copied from the cache
`.quak-backup-<fileID>.<ext>-<pid>-<random>.tmp`. A run that is killed can leave
one of these temporary files behind; the next backup deletes those whose process
is no longer running. The content cache uses the same scheme, and opening a
library deletes the temporary files in the cache whose process is no longer
running, so a download another process has in progress in the same cache is left
alone. The rename replaces whatever was at the destination rather than writing
through it: a symlink there is replaced, not followed, and the new file has the
temporary file's permissions, not those of the file it replaced.
appears in. On subsequent runs, existing originals are skipped. If a download
fails, the error is logged and the backup continues with the next file. The exit
code is non-zero if any files failed.
## TODO
- [x] Retry policy: no retry on 4xx, exponential backoff on 5xx and network
errors
- [x] Update the API reference section below to match the current implementation
- [ ] Update the API reference section below to match the current implementation
- [x] `make docker` green
- [ ] Store live photos in a form a photo viewer can open
(https://git.eeqj.de/sneak/quak/issues/107), once sneak has chosen between
keeping the ZIP and unpacking it
Tagging and releases are decided by sneak alone, and happen only when he
declares one.
- [ ] Tag `v1.0.0`
Future (desktop client, separate repo):
@@ -572,194 +474,25 @@ Future (desktop client, separate repo):
## API reference
The library's primary surface is the cache-backed `Library`; the lower-level
`Client` sits underneath it and is covered by the Design sections above. The
test suite is the canonical, executable documentation — `test/library/` and
`test/client/usage.test.ts` walk every operation, and `yarn test` verifies them.
The API reference section below is from an earlier draft and does not fully
reflect the current implementation. The authoritative API documentation is in
the test files, particularly `test/client/usage.test.ts` which is a literate
tutorial walking through every operation. Run `yarn test` to verify the examples
are correct.
### Opening a library
The key types and their actual signatures can be found in:
`Library.open(options)` loads the on-disk cache, starts the background refresh
loop, and resolves to a `Library`. On an empty cache it awaits the first
refresh, so it opens onto the account's data whenever the server is reachable;
if that refresh fails, it opens with no data and records the error in
`lib.status()`. On an existing cache it returns immediately and refreshes in the
background, so an unreachable server does not block opening.
`LibraryOptions`:
| Option | Default | Meaning |
| ------------------------ | --------------------------- | --------------------------------------------------------------------- |
| `client` | required | the account client (a `Client`, or any `LibraryClient`) |
| `cacheDirectory` | `<XDG cache>/quak/<userID>` | where `metadata.json` and the content cache live |
| `downloadDirectory` | none | backup destination; an original already stored there counts as cached |
| `refreshIntervalSeconds` | `3` | background refresh cadence |
| `precacheThumbnails` | `true` | prefetch every thumbnail, newest first |
| `precacheOriginals` | `true` | prefetch the favorites album and the latest-window originals |
| `precacheOriginalsDays` | `7` | length in days of that latest window |
| `cacheOriginalsMaxBytes` | 100 GiB | hard ceiling on the originals cache |
| `freeBelowBytes` | 50 GiB | free space to protect on the volume; the effective limit adapts down |
| `isOriginalPinned` | none | extra predicate for originals that must never be evicted |
| `pools` | fresh `RequestPools` | the bounded request pools (sets concurrency) |
| `onProgress` | none | refresh/ML/precache progress callback (`RefreshEvent`) |
| `contentSource` | the client's own | override the byte source (mainly for tests) |
Concurrency is set through `pools`: construct
`new RequestPools({ metadataConcurrency, contentConcurrency, thumbnailConcurrency })`
and pass it. The three pools default to 10 / 5 / 25 (see Request pools below).
`lib.status()` returns a `LibraryStatus` (collection/file counts, last
refresh/ML times and errors, originals usage and effective limit, precache
progress, and `closed`). `lib.close()` stops the background timer; it is
idempotent, and an in-flight refresh is left to finish. The promise it returns
resolves once that refresh (including its cache write), the ML data fetch and
the precache fetches already running have all finished, so the cache directory
can then be removed.
### Default reads vs. fresh reads
Default reads — `lib.albums`, `lib.photos`, `lib.timeline` — answer
synchronously from the last refreshed copy held in RAM and never touch the
network. The background timer refreshes that copy every
`refreshIntervalSeconds`, so a default read is immediate but may be up to one
interval stale.
`await lib.fresh()` forces a refresh, waits for it to complete and persist, and
returns the same `{ albums, photos, timeline }` namespaces — now guaranteed to
reflect a completed server round-trip. Concurrent `fresh()` calls coalesce onto
one refresh, and a refresh that fails rejects the caller (default reads stay
silent and keep serving the last good copy). The CLI's read commands use fresh
reads (issue https://git.eeqj.de/sneak/quak/issues/75).
### Read surface
- `lib.albums.list()``Album[]`, newest-updated first.
`lib.albums.byID({ collectionID })` and `byName({ albumName })`
`Album | undefined`.
- `lib.photos.byID({ fileID })``Photo | undefined`.
`lib.photos.records({ fileIDs })``PhotoRecord[]` in the requested order,
each id once, unknown ids dropped.
- `lib.timeline.groups({ groupBy, filter? })``TimelineGroup[]`, grouped by
`"day" | "week" | "month"` (keys `YYYY-MM-DD`, ISO `YYYY-Www`, `YYYY-MM`),
newest group first. A `PhotoFilter` combines `albumID`, `text`
(title/caption/album-name substring), `fileTypes`, `hasLocation`, and
`includeArchived`; hidden photos are always excluded.
An `Album` exposes its record fields and `album.photos.list()``Photo[]`
(newest first). A `Photo` exposes its record fields, `photo.record()`
`PhotoRecord`, and two content methods:
- `await photo.original(opts?)``{ path, bytes }` — the full-resolution file.
- `await photo.thumbnail(opts?)``{ path, bytes }`.
Both serve from the on-disk content cache when the bytes are present and
otherwise fetch through the pools; `opts.onProgress` reports per-file progress.
They throw when the library was opened without a content source.
Lower-level accessors that return decrypted model objects (which hold key
material) are also available: `listCollections()`, `getCollection(id)`,
`listFiles(collectionID)`, `getFile(collectionID, fileID)`, and
`getFileByID(fileID)`.
### Records and change notifications
The GUI-facing records hold no key material and no binary, so they survive
`structuredClone`/JSON across the Electron IPC boundary:
- `PhotoRecord`: `fileID`, `albumIDs`, `title`, `takenAt` (milliseconds),
`fileType`, optional `caption` / `width` / `height` / `latitude` /
`longitude`, `isArchived`, `isHidden`, and `thumbnailPath` / `originalPath`
once the bytes are cached.
- `AlbumRecord`: `collectionID`, `name`, `type`, `isShared`, `updationTime`, and
`fileIDs` (newest first).
- `LibrarySnapshot`: `{ albums, photos, takenAt }`.
`lib.snapshot()` returns a `LibrarySnapshot` (albums newest-updated first,
photos newest first). `lib.subscribe({ onChange })` delivers a `LibraryChange`
(`albumsChanged`, `photosChanged`, `fileIDsRemoved`, `albumIDsRemoved`,
`refreshedAt`) whenever a refresh alters the projection, and returns
`{ unsubscribe }`; a refresh that changes nothing delivers nothing.
### Thumbnails, ML search, and backup
- `lib.thumbnails.ensure({ fileIDs, priority, signal?, onProgress? })`
prefetches thumbnails through the thumbnail pool, deduped by fileID, returning
one `EnsureResult` (`{ fileID, path?, error? }`) per file. `priority` is
`"visible" | "ahead" | "background"`; only `"visible"` preempts background
work.
- `lib.mldata` searches the CLIP index built from Ente's per-file ML data:
`forFile({ fileID })``Promise<MLData | undefined>` (the whole stored
payload — face boxes, landmarks, embedding — read from disk on demand);
`similar({ fileID, limit? })` and `searchByEmbedding({ embedding, limit? })`
`SimilarResult[]` (`{ fileID, score }`, cosine similarity, most similar first,
default limit 20). quak bundles no text encoder, so `searchByEmbedding` takes
a query vector the caller produced elsewhere.
- `await lib.backup(opts?)``BackupResult`. It refreshes, fetches every
in-scope original not already in the backup (and, with `includeThumbnails`,
thumbnails) through the content cache, and rebuilds the on-disk backup tree
with a durable failure ledger. A fetched original is written straight into the
backup's `originals/` and not into the cache, which then counts it as present;
one the cache already held is copied from there. `BackupOptions`:
`downloadDirectory` (falls back to the one `open()` was given),
`includeOriginals` (default `true`), `includeThumbnails` (default `false`),
`onlyAlbumNames`, and `onProgress`. See Backup layout above for the tree it
writes.
### Request pools
`RequestPools` holds three independent bounded pools — metadata (10), content
(5), thumbnails (25) — because Ente meters these traffic classes differently.
Each pool orders on-demand work ahead of background/precache work and dedups
in-flight fetches by key, and an idle pool never lends its slots to a busy one.
### On-disk cache layout
Under `cacheDirectory`:
```
<cacheDirectory>/
metadata.json decrypted account state + refresh cursor
originals/<fileID>.<ext> cached full-resolution files
thumbnails/<fileID>.jpg cached thumbnails
mldata/
<fileID>.json one decrypted ML payload per file
clip.f32, clip.json the packed CLIP index and its id list
fetched.json per-file fetch bookkeeping
```
When `metadata.json` belongs to a different account than the client's,
`Library.open` deletes it and `mldata/` and starts from an empty cache. Cached
originals and thumbnails are kept; they are reached only through the files the
current account's records name.
A stored file appears only via an atomic temp-then-rename, so its presence means
it is complete. Every downloaded original (by `quak get`, the cache, or
`backup`) whose metadata records a content hash (`FileMetadata.hash`) is hashed
as it is written: unkeyed BLAKE2b with a 64-byte output, standard base64. For a
live photo, which is stored as a ZIP, the image and the video are hashed
separately and joined as `<imageHash>:<videoHash>`. A mismatch stores nothing
and fails the download with an error naming the file ID. An original with no
recorded hash, from a very old client, is stored unchecked.
### Key types by source file
- `src/library/index.ts`: `Library`, `LibraryOptions`, `LibraryStatus`,
`LibraryClient`, `RefreshEvent`
- `src/library/read.ts`: `Album`, `Photo`, `AlbumsAPI`, `PhotosAPI`,
`TimelineAPI`, `PhotoFilter`, `TimelineGroup`, `GroupBy`
- `src/library/content.ts`: `ContentResult`, `ContentOptions`, `ThumbnailsAPI`,
`EnsureOptions`, `EnsureResult`, `ContentSource`
- `src/library/records.ts`: `PhotoRecord`, `AlbumRecord`, `LibrarySnapshot`,
`LibraryChange`
- `src/library/mlsearch.ts`: `MLDataAPI`, `SimilarResult`
- `src/library/pools.ts`: `RequestPools`, `RequestPoolsOptions`, `BoundedPool`
- `src/backup.ts`: `BackupOptions`, `BackupResult`, `BackupError`
- `src/client.ts`: `Client`, `LoginOptions`, `ClientSnapshot`
- `src/api/client.ts`: `ApiClient`, `ApiClientOptions`, `StreamOptions`
- `src/api/client.ts`: `ApiClient`, `ApiClientOptions`, `ApiError`,
`StreamOptions`
- `src/errors.ts`: `ApiError`, `TruncatedStreamError`
- `src/retry.ts`: `withRetry`, `isRetryable`, `isSafeToReplay`, `RetryOptions`
- `src/model/types.ts`: `Collection`, `EnteFile`, `FileMetadata`, `FileType`,
`CollectionType`, `RawCollection`, `RawEnteFile`
- `src/auth/types.ts`: `KeyAttributes`, `SRPAttributes`,
`AuthorizationResponse`, `LoginChallenge`
- `src/model/types.ts`: `Collection`, `EnteFile`, `FileMetadata`, `FileBlob`,
`RawCollection`, `RawEnteFile`, `RawMagicMetadata`
- `src/download/index.ts`: `DownloadResult`
- `src/backup.ts`: `BackupResult`, `BackupError`
- `src/thumbnails.ts`: `MissingThumbnailInfo`, `ThumbnailFixResult`
## Source attribution
@@ -788,22 +521,20 @@ documents:
commented thoroughly. `main` is always green.
- **Required checks before every commit:** `make lint` must pass — that is
eslint plus the prettier check, and it builds the `lint` phase of the
`Dockerfile`, so it needs docker. The pre-commit hook enforces exactly that.
`make check` (which also runs the tests) must pass before merging to `main`.
`make fmt-check` is available for a host-side formatting check on its own, but
it is not a separate requirement: `make lint` already covers it, and running
both would check formatting twice. Never invoke eslint or prettier directly;
linting runs in the container only.
eslint plus the prettier check, and it builds `Dockerfile.lint`, so it needs
docker. The pre-commit hook enforces exactly that. `make check` (which also
runs the tests) must pass before merging to `main`. `make fmt-check` is
available for a host-side formatting check on its own, but it is not a
separate requirement: `make lint` already covers it, and running both would
check formatting twice. Never invoke eslint or prettier directly; linting runs
in the container only.
- **Formatting:** prettier with 4-space indents and `proseWrap: always` for
markdown. Use `make fmt` to format. Use `yarn` not `npm`.
- **Testing:** vitest. Tests go in `test/` mirroring the `src/` structure.
`make test` must finish in under 60 seconds (the hard cap) and should finish
in under 20. The 90-second `timeout` in the `test` phase of the `Dockerfile`
is a backstop that catches a hung test, not the time limit. Use `mkdtempSync`
for temporary directories, never manual timestamp paths.
`make test` must complete in under 20 seconds. Use `mkdtempSync` for temporary
directories, never manual timestamp paths.
- **Code style:** `const` for everything, `let` if reassignment is needed, never
`var`. Avoid unnecessary comments. No hand-rolled crypto. The
+75 -270
View File
@@ -1,6 +1,6 @@
---
title: Repository Policies
last_modified: 2026-09-08
last_modified: 2026-07-06
---
This document covers repository structure, tooling, and workflow standards. Code
@@ -60,28 +60,17 @@ style conventions are in separate documents:
prerequisite since nvm requires bash. yarn is then pinned via
`corepack prepare yarn@<version> --activate`. Never install "latest" or "lts";
always exact versions. `script/cibuild` runs the CI build: it changes to the
repo root, runs `script/bootstrap`, runs `script/check`, and builds the image
with the version; the Gitea workflow calls it. **`script/cibuild` runs
`script/bootstrap` first**, because the workflow checks out the repo and runs
nothing else, while `script/fmt-check` runs the formatter on the host: on a
pristine checkout with nothing installed the run dies there, after the
containerised gates have passed. **The bootstrap alone is not enough**:
`script/bootstrap` installs node and yarn under nvm and leaves neither on the
`PATH` of the shell that called it, so a bare `yarn` still exits 127. The host
entrypoints that need yarn — `script/fmt` and `script/fmt-check` — therefore
source nvm for the pinned node version before invoking it, exactly as
`script/bootstrap`'s own install step does. A runner carrying nothing but
docker and git then gets through `script/check`. Four further scripts are our
own extensions to the standard: `script/check` runs `script/test`,
`script/lint` and `script/fmt-check`; `script/precommit` is what the git
pre-commit hook runs, and it calls `script/check`; `script/install-precommit`
installs the git pre-commit hook (the `make hooks` target shims to it); and
`script/projectname` (literally that filename) simply outputs the project's
name. Scripts that need the name call `script/projectname` — e.g.
`script/docker` assembles its image tag from it — so those scripts stay
byte-identical across all repos. Repo-type-specific pre-commit extras (e.g.
`go mod tidy` verification in Go repos) belong in `script/precommit`, not in
the hook itself. Model scripts are at
repo root and runs `docker build .`; the Gitea workflow calls it. Four further
scripts are our own extensions to the standard: `script/check` runs
`script/test`, `script/lint`, and `script/fmt-check`; `script/precommit` is
what the git pre-commit hook runs, and it calls `script/check`;
`script/install-precommit` installs the git pre-commit hook (the `make hooks`
target shims to it); and `script/projectname` (literally that filename) simply
outputs the project's name. Scripts that need the name call
`script/projectname` — e.g. `script/docker` assembles its image tag from it —
so those scripts stay byte-identical across all repos. Repo-type-specific
pre-commit extras (e.g. `go mod tidy` verification in Go repos) belong in
`script/precommit`, not in the hook itself. Model scripts are at
`https://git.eeqj.de/sneak/prompts/raw/branch/main/script/<name>`. The README
must document the provided scripts in an **Entrypoints** section (see the
README requirements below).
@@ -100,140 +89,87 @@ style conventions are in separate documents:
contributor should be able to understand the entire development workflow by
reading the Makefile.
- Every repo should have a `Dockerfile`, and it carries the repo's gates: a
`lint` phase and a `test` phase, with the final stage depending on both so the
image cannot be built unless they pass. For non-server repos the final stage
brings up a development environment; for server repos it is the runtime image.
Dockerfiles install development prerequisites by running `script/bootstrap`
rather than duplicating installs inline; COPY `script/` and the dependency
manifests (`package.json` + `yarn.lock`, `go.mod` + `go.sum`, etc.) before
running it.
- Every repo should have a `Dockerfile`. All Dockerfiles must run `make check`
as a build step so the build fails if the branch is not green. For non-server
repos, the Dockerfile should bring up a development environment and run
`make check`. For server repos, `make check` should run as an early build
stage before the final image is assembled. Dockerfiles install development
prerequisites by running `script/bootstrap` rather than duplicating installs
inline; COPY `script/` and the dependency manifests (`package.json` +
`yarn.lock`, `go.mod` + `go.sum`, etc.) before running it so the bootstrap
layer stays cached until dependencies change.
- **Linting and testing run in Docker, as phases of the `Dockerfile`.** There is
no separate lint file. `script/lint` and `script/test` each build one phase
and nothing else:
- **Dockerfiles must use a separate lint stage for fail-fast feedback.** Go
repos use a multistage build where linting runs in an independent stage based
on the `golangci/golangci-lint` image (pinned by hash). This stage runs
`make fmt-check` and `make lint` before the full build begins. The build stage
then declares an explicit dependency on the lint stage via
`COPY --from=lint /src/go.sum /dev/null`, which forces BuildKit to complete
linting before proceeding to compilation and tests. This ensures lint failures
surface in seconds rather than minutes, without blocking on dependency
download or compilation in the build stage.
```sh
docker build --no-cache --target lint -t "$(script/projectname)-lint" .
docker build --no-cache --target test -t "$(script/projectname)-test" .
```
**A stage that is not the last one in the file is built only when the final
stage's chain depends on it, or when `--target` names it.** That is why the
two gates are always invoked by name here, and why the final stage carries a
`COPY --from=` of a harmless file from each of them: without that edge a
plain `docker build .` builds the last stage alone and exits 0 having linted
and tested nothing.
**Every `docker build` in `script/` is tagged**, here and in
`script/cibuild` and `script/docker`. An untagged build leaves a dangling
image behind on every invocation, on every developer host and every CI
runner; a tagged one replaces the previous image.
Inside a phase the tool is invoked directly — `golangci-lint`, `go test`,
`eslint`, `prettier` — never through `make lint` or `script/test`, which are
themselves a `docker build` and would recurse into a daemon that does not
exist in a build step. Formatting is the exception and stays on the host:
`script/fmt` writes the working tree, and `script/fmt-check` is its
read-only twin.
**No lint verdict may come from a host invocation of the linter.** On a
shared host golangci-lint reads a result cache keyed on file content rather
than location, so a second checkout of the same content is served the first
one's findings, and a host-global lock in `$TMPDIR` makes concurrent runs
exit non-zero with `parallel golangci-lint is running` — a status a caller
cannot tell from real findings. Both have produced wrong verdicts in this
org, in both directions. A container has its own cache, its own `TMPDIR` and
a digest-pinned binary, so neither is reachable.
- **Any build that runs checks is built with `--no-cache`.** Docker invalidates
a `COPY` layer only when the copied content changes, so on an unchanged tree
the check `RUN` is served from cache, nothing executes, and the build still
exits 0. Every `docker build` in `script/` therefore passes `--no-cache`:
`script/lint`, `script/test`, `script/cibuild` and `script/docker` are the
four, and there is no fifth — `script/check` runs the two gate phases and
`script/fmt-check`, and builds no image of its own. A bare `docker build .` is
not evidence that anything ran: a sub-second build reporting success is a
cache hit, not a result. Never invalidate by pruning — `docker builder prune`
and friends destroy a build cache shared with every other build on the host.
- **The gate phases are separate stages, and the build stage depends on both.**
The lint phase is based on the `golangci/golangci-lint` image (pinned by
hash), so lint failures surface in seconds rather than after a full compile,
and the test phase is based on the Go image. The canonical Go repo
`Dockerfile`:
The standard pattern for a Go repo Dockerfile is:
```dockerfile
# Lint phase
# Lint stage — fast feedback on formatting and lint issues
# golangci/golangci-lint:v2.x.x, YYYY-MM-DD
FROM golangci/golangci-lint@sha256:... AS lint
WORKDIR /src
COPY go.mod go.sum ./
RUN go mod download
COPY . .
RUN golangci-lint run --config .golangci.yml ./...
RUN make fmt-check
RUN make lint
# Test phase
# golang:1.x-alpine, YYYY-MM-DD
FROM golang@sha256:... AS test
WORKDIR /src
COPY go.mod go.sum ./
RUN go mod download
COPY . .
RUN go test -timeout 90s -race -cover ./... || \
{ echo "--- Rerunning with -v for details ---"; \
go test -timeout 90s -race -v ./...; exit 1; }
# Build stage. Nothing is wanted from either phase above; the copies
# are what make BuildKit build them first, so this stage cannot run
# unless lint and test passed.
# Build stage
# golang:1.x-alpine, YYYY-MM-DD
FROM golang@sha256:... AS builder
COPY --from=lint /src/go.sum /dev/null
COPY --from=test /src/go.sum /dev/null
WORKDIR /src
# Force BuildKit to run the lint stage before proceeding
COPY --from=lint /src/go.sum /dev/null
COPY go.mod go.sum ./
RUN go mod download
COPY . .
RUN make test
ARG VERSION=dev
RUN CGO_ENABLED=0 go build -trimpath \
-ldflags="-s -w -X main.Version=${VERSION}" \
-o /app ./cmd/app/
# Runtime stage, and the last one
# Runtime stage
FROM alpine@sha256:...
COPY --from=builder /app /usr/local/bin/app
ENTRYPOINT ["app"]
```
Key points:
- The lint phase uses the `golangci/golangci-lint` image directly (it has
both Go and the linter), so nothing needs installing.
- `COPY --from=<phase> /src/go.sum /dev/null` is a no-op copy whose only
purpose is the ordering edge. BuildKit runs stages in parallel by default,
and a stage nothing depends on is not built at all, so without these two
lines a red gate would not fail the build.
- Keep the runtime stage last, and if you add a stage after it, give it the
same two copies. A plain `docker build .` builds the last stage's chain
and nothing else.
- The lint stage uses the `golangci/golangci-lint` image directly (it
includes both Go and the linter), so there is no need to install the
linter separately.
- `COPY --from=lint /src/go.sum /dev/null` is a no-op file copy that creates
a stage dependency. BuildKit runs stages in parallel by default; without
this line, the build stage would not wait for lint to finish and a lint
failure might not fail the overall build.
- If the project uses `//go:embed` directives that reference build artifacts
(e.g. a web frontend compiled in a separate stage), the lint phase must
(e.g. a web frontend compiled in a separate stage), the lint stage must
create placeholder files so the embed directives resolve. Example:
`RUN mkdir -p web/dist && touch web/dist/index.html web/dist/style.css`.
The lint stage should not depend on the actual build output — it exists to
fail fast.
- If the project requires CGO or system libraries for linting (e.g.
`vips-dev`), install them in the lint phase with `apk add`.
- `ARG VERSION=dev` is declared in the stage that compiles and supplied by
`script/docker` and `script/cibuild`; no stage may call `git describe`.
`vips-dev`), install them in the lint stage with `apk add`.
- The build stage runs `make test` after compilation setup. Tests run in the
build stage, not the lint stage, because they may require compiled
artifacts or heavier dependencies.
- Every repo should have a Gitea Actions workflow (`.gitea/workflows/`) that
runs `script/cibuild` on push, and checks out the repo as its only other step.
That script bootstraps, runs the gate phases, and then builds the image, so a
successful run means every check passed; a bare `docker build .` does not
carry the same guarantee, because its gate phases may come from the cache. The
image build is uncached and so runs the gate phases a second time. That is the
price of the rule above, and it is worth paying: the image that ships is built
from a run of its own gates rather than from a cache entry.
runs `script/cibuild` (which runs `docker build .`) on push. Since the
Dockerfile already runs `make check`, a successful build implies all checks
pass.
- Use platform-standard formatters: `black` for Python, `prettier` for
JS/CSS/Markdown/HTML, `go fmt` for Go. Always use default configuration with
@@ -253,21 +189,14 @@ style conventions are in separate documents:
module under test to verify it compiles/parses. There is no excuse for
`make test` to be a no-op.
- `make test` must complete in under 60 seconds. That is the hard cap, and a
suite that exceeds it fails. Under 20 seconds is the target. A suite between
20 and 60 seconds is still green, but the overage must be filed as an
improvement bug against that repo. Add a 90-second timeout to the test
invocation (`go test -timeout 90s`). The backstop deliberately sits above the
hard cap so that it catches a genuinely hung test rather than a merely slow
one.
- `make test` must complete in under 20 seconds. Add a 30-second timeout in the
Makefile.
- **The test command should use the conditional verbose rerun pattern.** Run
tests without `-v` (verbose) first. If tests fail, automatically rerun with
`-v` to show full output. This keeps CI logs and `docker build` output clean
on success (just package/suite summaries) while providing full diagnostic
detail on failure (every test case, every assertion). The command lives in the
`test` phase of the `Dockerfile`, since `script/test` builds that phase; the
Makefile form below is the same pattern for any repo-local invocation:
- **`make test` should use the conditional verbose rerun pattern.** Run tests
without `-v` (verbose) first. If tests fail, automatically rerun with `-v` to
show full output. This keeps CI logs and `docker build` output clean on
success (just package/suite summaries) while providing full diagnostic detail
on failure (every test case, every assertion). The general shell pattern:
```makefile
test:
@@ -280,24 +209,11 @@ style conventions are in separate documents:
```makefile
test:
@go test -count=1 -timeout 90s -race -cover ./... || \
@go test -timeout 30s -race -cover ./... || \
{ echo "--- Rerunning with -v for details ---"; \
go test -count=1 -timeout 90s -race -v ./...; exit 1; }
go test -timeout 30s -race -v ./...; exit 1; }
```
`-count=1` is required on both invocations: it defeats Go's test _result_
cache, so the target cannot report a pass it did not earn, and the rerun
reproduces a failure instead of replaying it. It leaves the build cache
alone, so it costs the runtime of the suite and no recompilation.
Note that this is a second, independent cache, stacked below the Docker
layer cache that [issue #26](https://git.eeqj.de/sneak/prompts/issues/26)
addresses. `CHECK_EPOCH` guarantees the `RUN make test` _step_ re-executes;
it does not guarantee `go test` inside that step does any work, because the
`GOCACHE` baked into earlier image layers survives into the re-executed
step. They are two separate defects requiring two separate fixes, and a fix
for one must not be recorded as covering the other.
Python example:
```makefile
@@ -323,83 +239,10 @@ style conventions are in separate documents:
must be in `.gitignore`. No exceptions.
- `.gitignore` should be comprehensive from the start: OS files (`.DS_Store`),
editor files (`.swp`, `*~`), in-repo agent scratch directories (`.claude/`),
language build artifacts, and `node_modules/`. Fetch the standard `.gitignore`
from `https://git.eeqj.de/sneak/prompts/raw/branch/main/.gitignore` when
setting up a new repo. These patterns are written to `.gitignore`'s own
semantics, in which an unanchored pattern already matches at every depth; they
are not a `.dockerignore` and must not be transplanted into one unmodified.
- **`.dockerignore` does not use `.gitignore` semantics, and copying patterns
across unmodified leaves secrets in the build context.** Docker matches with
`moby/patternmatcher`: `filepath.Match` semantics plus a `**` extension, so
`*` does not cross `/` and a pattern without a leading `**/` is anchored at
the build-context root. A `.dockerignore` listing `.env`, `*.pem` and `*.key`
therefore excludes only the copies at the repository root, while `config/.env`
and `certs/server.key` still reach the context and can land in an image layer
— which is more dangerous than a short file with no secret patterns at all,
because it reads as solved and stops anyone looking. Give every
depth-independent pattern the `**/` prefix and leave only genuinely
root-anchored entries unprefixed: `.git`, and the repo's own host-built
binary, written `/myapp` and never `**/myapp`, which would also match
`cmd/myapp/` and delete the package directory from the context. Matching is
case-sensitive, and an ALL-CAPS twin per pattern still misses `Server.Key`, so
secret names use character ranges — `**/*.[kK][eE][yY]`, `**/*.[pP][eE][mM]`,
and likewise for `.envrc` and the extensionless SSH keys. Where such a pattern
also catches something the build needs, re-include it with a negation
(`!docs/example.env`); deleting the pattern reopens the exposure for every
other file it covers. Fetch the standard `.dockerignore` from
`https://git.eeqj.de/sneak/prompts/raw/branch/main/.dockerignore` and extend
it with the repo's own artifacts.
- **In-repo agent scratch belongs in both files, written to each file's own
semantics.** `.claude/` holds one worktree per in-flight agent — an entire
additional checkout of the repo — so under `COPY . .` the build context
inflates by a multiple of the repo and another session's unreviewed work can
be copied into an image layer. In `.gitignore` the entry is `.claude/`,
unanchored. In `.dockerignore` it is `.claude`, anchored and with **no** `**/`
prefix, because the prefixed form would also delete any nested directory of
that name from the build. Anchoring carries a known gap that the canonical
`.dockerignore` states in its own comment, since consuming repos receive the
file and not the tracker: the directory is created in the agent's working
directory, so a repo running agents in subdirectories still ships
`services/api/.claude/` and must add its own anchored entry there.
- **Excluding `.git` means `git describe` cannot run inside any build stage, and
it fails quietly there.** In a build stage there is no repository, so
`git describe` writes nothing to stdout, `-X main.Version=` comes out empty,
the binary reports no version at all, and the build still exits 0. Compute the
version on the host and thread it in as a build arg. `script/docker` and
`script/cibuild` do this, byte-identically across repos:
```sh
# Own line: a failing command substitution inside an argument does not
# trip `set -e`, so the inline form degrades to an empty constant.
version="$(git describe --tags --always --dirty 2>/dev/null || true)"
[ -n "$version" ] || version="unknown"
docker build --no-cache \
--build-arg VERSION="$version" \
-t "$(script/projectname)" .
```
`--always` makes an untagged repo yield an abbreviated commit hash rather
than failing, and the `[ -n "$version" ]` line is the single place the
fallback is applied — a live check that fires on a build from an export with
no `.git` and on a repository with no commits yet. Do not fold it into the
substitution as `|| echo unknown`, which makes the guard unreachable. The
Dockerfile's side is `ARG VERSION=dev` in the stage that compiles, declared
there because `ARG` is stage-scoped; passing `VERSION` to a repo whose
Dockerfile declares no such `ARG` is ignored and costs nothing, which is why
the scripts stay byte-identical. One consequence for CI: the standard
checkout action clones shallow and fetches no tags, so a repo that embeds a
tag-derived version must set `fetch-depth: 0` on its checkout step.
- **Verify `.dockerignore` by enumerating the image, not by reading the
patterns.** Plant files at the root _and_ at least two directories deep, build
a probe image that does `COPY . .`, and list what actually landed
(`docker run --rm --entrypoint find IMAGE /app`). The `transferring context`
size is not a substitute: a nested secret is a few bytes, and BuildKit
transfers only the delta from the previous build.
editor files (`.swp`, `*~`), language build artifacts, and `node_modules/`.
Fetch the standard `.gitignore` from
`https://git.eeqj.de/sneak/prompts/raw/branch/main/.gitignore` when setting up
a new repo.
- **No build artifacts in version control.** Code-derived data (compiled
bundles, minified output, generated assets) must never be committed to the
@@ -415,45 +258,9 @@ style conventions are in separate documents:
- Make all changes on a feature branch. You can do whatever you want on a
feature branch.
- `.golangci.yml` is standardized. The vendored copy in a consuming repo must
_NEVER_ be modified by an agent: fetch it from
`https://git.eeqj.de/sneak/prompts/raw/branch/main/.golangci.yml` and keep it
byte-identical, so that no repo can quietly loosen its own linting. Linter
configuration changes are made to the canonical copy in the `prompts` repo and
reach consuming repos by re-vendoring; an agent may open a PR against
canonical, which only the user merges. One list is exempt from byte-identity,
because it cannot be written once for every repo: the `deny` list of the
`test-support` depguard rule, where a repo names its own test-support packages
by full import path. A repo adds entries there and changes nothing else, and a
re-vendor carries its entries forward. The canonical golangci-lint version is
v2.12.2 (released 2026-05-06), pinned as the digest of the lint phase's base
image
(`golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240`,
which reports `2.12.2 built with go1.26.2 from c0d3ddc9`). That digest is the
only pin, since no repo installs golangci-lint on the host: bumping the
version means changing it and nothing else.
- **`script/bootstrap` installs a pinned tool by comparing versions, never by
testing presence.** An `if ! command -v <tool>; then install; fi` guard tests
`PATH` only, so on an already-provisioned machine the pin is inert and a
version bump is a silent no-op — while the Dockerfile, installing into a clean
image, gets the pinned version, so a local `make check` and `make docker` can
disagree about what the tool even is. The canonical form:
- compares the installed version against the pin over the **whole** version
token; a parser that stops at the first `-` reports `2.12.2` for a host
running `2.12.2-rc1` and skips the install;
- treats absent, non-zero, empty or unrecognised `--version` output as a
mismatch, so the failure direction is a redundant install and never a
skipped one;
- after installing, re-resolves the binary the way callers do — `hash -r`,
then through `PATH`, not through the directory the installer wrote to —
and fails naming the resolved path, since an install that a shadowing
binary hides succeeds while changing nothing any caller sees;
- is actually called, and prints the version on both success paths: a
function defined and never invoked has the same exit status and the same
empty output as one that worked.
Keep it POSIX sh: no arrays, no `[[`, no `grep -P`.
- `.golangci.yml` is standardized and must _NEVER_ be modified by an agent, only
manually by the user. Fetch from
`https://git.eeqj.de/sneak/prompts/raw/branch/main/.golangci.yml`.
- When pinning images or packages by hash, add a comment above the reference
with the version and date (YYYY-MM-DD).
@@ -572,9 +379,7 @@ style conventions are in separate documents:
language-specific config). Everything else goes in a subdirectory. Canonical
subdirectory names:
- `bin/` — executable scripts and tools
- `cmd/` — Go command entrypoints; thin only: one `main.go` per binary whose
body is a single call into `internal/` or `pkg/`, no project logic in
`cmd/`
- `cmd/` — Go command entrypoints
- `configs/` — configuration templates and examples
- `deploy/` — deployment manifests (k8s, compose, terraform)
- `docs/` — documentation and markdown (README.md stays in root)
+2 -214
View File
@@ -14,223 +14,10 @@ pre-1.0
# Next Step
Store live photos in a form a photo viewer can open
(https://git.eeqj.de/sneak/quak/issues/107). This waits on sneak's choice
between keeping the ZIP and unpacking it into the image and the video.
Tagging and releases are decided by sneak alone, and happen only when he
declares one.
Update the README API reference section to match the current implementation.
# Completed Steps
- 2026-09-23: Settled the package metadata (issue 6). quak is not published, so
`package.json` is marked `"private": true` and the `files` field is gone.
`engines.node` is `>=22`, the major version `script/bootstrap` and the
`Dockerfile` use. An `exports` map makes `.` and `./package.json` the only
importable paths; `runMetadataBackup`, the thumbnail helpers and their types
stay internal to the CLI.
- 2026-09-23: Brought the README and this file in line with the tree (issue
111). The layout lists `src/library/` and the other source files, the backup
layout names `failures.json` and the optional `thumbnails/`, "Opening a
library" says what happens when the first refresh fails, the Testing section
gives the 60-second hard cap and 20-second target for `make test` and names
the 90-second `timeout` in the `Dockerfile` as the backstop for a hung test,
and "Tag v1.0.0" is no longer listed as the next step.
- 2026-09-23: Tested `quak login` and `backup-metadata --exif` (issue 110).
`loginCommand` takes its login function and its prompts from `CliContext`, and
`bin/quak.ts` passes `Client.login` and the terminal prompts. Tests cover a
login from `QUAK_EMAIL` and `QUAK_PASSWORD` with no prompt, the TOTP prompt, a
failed login, and the saved session's modes, and show that `--exif` and
`--all` each turn on EXIF extraction and that it is off without them.
- 2026-09-23: `quak backup` writes each original once and no longer fills the
cache (issue 106). An original fetched for a backup is written by the download
writer straight into the backup's `originals/`, and the content cache records
it there instead of keeping its own copy; one the cache already held is still
copied. `quak backup` opens its library with the thumbnail and originals
precache off.
- 2026-09-23: `backup-metadata`, `helper list-missing-thumbnails` and
`helper fix-missing-thumbnails` refresh before they answer (issue 100). Each
awaits `lib.fresh()` before reading, so a file added since the cache was
written is included, and a failed refresh prints one line and exits 1 instead
of answering from a stale or empty cache. The README lists them among the
commands that refresh first.
- 2026-09-23: `quak backup` waits for the server refresh and fails when it fails
(issue 99). `lib.backup()` joins a refresh already running or starts one, as
`fresh()` does, and rejects before touching any file when it fails, leaving
`failures.json` as it was, so `quak backup` prints the error as one line and
exits 1 instead of backing up the previous run's file list, or nothing, and
exiting 0.
- 2026-09-23: CLI errors print a message instead of a stack trace (issue 102).
An error a command throws is printed as one `quak: MESSAGE` line on stderr and
the CLI exits 1 once output has drained. The wrapper that does this moved from
`bin/quak.ts` to `src/cli-run.ts`, and `bin/quak.ts` now awaits
`program.parseAsync()`.
- 2026-09-23: Opening a library no longer deletes another process's download in
progress (issue 105). The download writer's temp files are named
`.quak-<pid>-<random>.tmp`, and `removeLeftoverTempFiles`, moved from the
backup into the download module, deletes a `.quak-*.tmp` file only when the
process ID in its name is no longer running. The content cache calls it at
`open()` for `originals/` and `thumbnails/`, the backup as before.
- 2026-09-23: Re-vendored the lint and test setup from the template (issue 96).
Linting and testing are the `lint` and `test` phases of the `Dockerfile`;
`script/lint` and `script/test` each build one with `--no-cache`, and the last
stage compiles and depends on both, so `script/cibuild` is one build.
`Dockerfile.lint`, `CHECK_EPOCH`, `LINT_EPOCH` and the tests that checked them
are gone; `REPO_POLICIES.md` is re-copied.
- 2026-09-23: Stopped `helper fix-missing-thumbnails` retrying files the server
always refuses (issue 109). Both thumbnail helpers skip a file another account
owns without fetching it. The fixer skips a file whose recorded thumbnail size
is 0 or unknown before downloading it, and otherwise tries smaller encodings
(720 px quality 50 down to 160 px quality 20) until the encrypted thumbnail is
no larger than that size, skipping the file if none fits.
- 2026-09-23: Tested the live-photo hash check's error paths (issue 117). Tests
download a live photo whose ZIP names an unknown compression method, one whose
ZIP has no image entry and one with no video entry, and check that nothing is
stored and the error names the file ID; the unreadable one is not retried.
- 2026-09-23: `quak logout` ends the session on the server (issue 108). It calls
`POST /users/logout` through the new `Client.logoutOnServer()`, then deletes
`session.json` even when that call fails, says so and exits 1. It prints the
account's cache directory and says it still holds decrypted data. The default
cache path is now `defaultCacheDirectory()` in the library, shared with
`Library.open`.
- 2026-09-23: Fixed the backup's per-collection folders (issue 103). Two files
in one collection with the same title, and two collections with the same name,
each get their ID added to the name (`IMG_0001 (12345).JPG`, `Trip (10)/`), so
none replaces another's symlink or JSON. Each run removes symlinks into
`originals/` for files no longer in the collection, and the folders of deleted
or renamed collections, leaving anything else in `collections/` alone. The
README backup layout states the naming rule.
- 2026-09-23: Checked downloaded originals against their recorded content hash
(issue 68). `downloadFile`, which `quak get`, the content cache and backup all
use, hashes the decrypted bytes (unkeyed BLAKE2b-512, standard base64) and
stores nothing on a mismatch, failing with an error naming the file ID. A live
photo ZIP is unpacked as it streams with `fflate` and its image and video
hashed separately as `<imageHash>:<videoHash>`. `decryptFile` reads older
clients' `imageHash` and `videoHash` fields for live photos. A file with no
recorded hash is stored unchecked.
- 2026-09-23: Kept one account's cache from mixing with another's (issue 104).
When `metadata.json` in the cache directory was written for a different,
non-zero user ID than the client's, `Library.open` deletes it and `mldata/`
and starts empty, so the first refresh enumerates from 0. This only happens
with `--cache-dir` or an explicit `cacheDirectory`; the default path already
includes the user ID. A test opens one account's cache as another account.
- 2026-09-23: `backup-metadata` no longer stops on one failed ML data request
(issue 101). Each request of up to 200 files is tried on its own; a failed one
is logged, its files are written with the reason in `mlDataError`, and the
command exits 1 once the dump is complete. `fetchMLData`, which only this
command used, is gone; the command calls `fetchMLDataBatch` per batch.
- 2026-09-23: Single-sourced the version string (issue 5). `package.json` is the
only place it is written: `src/index.ts` imports it for `VERSION` and
`bin/quak.ts` passes `VERSION` to commander. tsc copies `package.json` to
`dist/package.json`, so the import resolves from the built output too, and
`script/build` runs the built CLI with `--version` to prove it. A test checks
that `VERSION` and `quak --version` both equal the `package.json` version.
- 2026-09-23: Tested that `Library.close()` waits for the originals precache
(issue 93). The test that holds a precache fetch open while `close()` runs now
runs once with only the thumbnail fill and once with only the originals fill,
so dropping either wait from `Precache.close()` fails a test.
- 2026-09-23: Fixed two intermittently failing library tests (issue 90).
`Library.close()` now returns a promise that resolves once an in-flight
refresh (including its cache write), the ML data fetch and running precache
sweeps have finished; the library tests await it, so `afterEach` no longer
removes the cache directory while something is still writing into it. The
precache test waits for both fills to report "done" instead of for its stub
source to be called, which happened before the cache recorded the file.
- 2026-09-23: Pinned three guards reviewers found untested (issue 89). The
download idle deadline's timer is unref'd, so it can never keep the process
alive, and a test checks no timer is left after a download completes or fails.
A test covers the rejection of `#` in a request path. The EXIF scan compares
the `Exif` header only in an APP1 segment of length 8 or more, so it never
reads the next segment's bytes, with a test for a short one.
- 2026-09-23: Made the download deadline an idle deadline (issue 24).
`downloadTimeoutMs` now aborts a file or thumbnail download only after no
bytes have arrived for that long, default 60 seconds, instead of bounding the
whole transfer at 10 minutes, so a slow download that keeps making progress
completes. A download that fails before reading the whole body cancels it, so
a failed file no longer holds its connection.
- 2026-09-23: Every `ApiClient` request URL is now built by one function next to
the class (issue 18), so a self-hosted `apiOrigin` with a base path keeps it
on every request, a path works with or without a leading slash, and query
parameters are percent-encoded. A path containing `?` or `#` is rejected with
an error instead of being silently cut.
- 2026-09-23: Made the CLI testable and tested it (issue 12). The command bodies
moved from `bin/quak.ts` into `src/cli-commands.ts` as functions that take
their options and a context (output streams, session directory, cache
directory, session loader) and return an exit code; `bin/quak.ts` only wires
them to commander and exits with the code once stdout and stderr have drained,
so nothing below it calls `process.exit`. `test/cli/commands.test.ts` drives
them with a fake client: session file modes, logout, the missing and corrupt
session paths, and the output and exit code of `whoami`, `collections`,
`files`, `get`, `get-thumb`, `backup` and `helper list-missing-thumbnails`.
- 2026-09-23: Hardened the backup tree's atomic copy (issue 22). `copyAtomic`
fsyncs its temp file before the rename and the directory after it, through the
download writer's `fsyncPath`; each backup run deletes `.quak-backup-*.tmp`
files whose process is no longer running. The README backup layout names the
temp files and states that the rename replaces a symlink and takes the temp
file's permissions. Added tests for a missing and an unwritable destination
directory for `downloadFile` and `downloadThumbnail`.
- 2026-09-23: Hardened the JPEG EXIF scan behind `backup-metadata --exif` (issue
11). Every segment length is checked against the remaining bytes and lengths
under 2 stop the scan, so a truncated or corrupt original can neither throw
nor loop. A malformed or unparseable EXIF segment is recorded as
`imageMetadata.exifError`, and a failure to read the original as
`imageMetadataError` in the per-file JSON, instead of the field being left
out.
- 2026-09-22: Hardened the retry classifier (issue 80). A `POST` or `PUT` is
replayed only when every errno in the cause chain is a connect errno, and it
no longer follows redirects. `getRetryOptions()` returns a copy. Tests pin
every errno the classifier names, the cause-chain depth limit, cycle
termination, and a fresh deadline per attempt for every retrying entry point.
The README's endpoint list is the one place that names the requests the replay
rule covers.
- 2026-09-22: Stopped `make test` collecting tests from checkouts nested under
`.claude/` (issue 25). vitest ignores `.gitignore` when finding tests, so a
nested checkout ran the whole suite again; `vitest.config.ts` now adds
`.claude/**` to vitest's default excludes, and
`test/packaging/nested-checkout.test.ts` plants a nested checkout in a temp
directory and fails if vitest would collect it.
- 2026-09-22: Dropped the deprecated `@types/libsodium-wrappers-sumo` stub from
`devDependencies` (issue 27). It shipped no declarations; the types come from
`libsodium-wrappers-sumo` itself. `yarn.lock` regenerated by `yarn remove`.
- 2026-09-22: Hardened the client session lifecycle (issue 10).
`Client.fromJSON` checks every snapshot field and each key's decoded length
and names the bad field; `toJSON` reads the token through
`ApiClient.getAuthToken` and throws when there is none; `logout` zeroes the
key buffers, and `collectionsSince` re-checks for logout after its request so
it never decrypts with zeroed keys. The CLI reports a corrupt session file
separately from a missing one (`src/cli-session.ts`).
- 2026-09-22: Sanitized file names taken from server metadata (issue 9). A new
`src/filename.ts` holds the one sanitizer, used by `quak get`/`get-thumb`
without `--out`, `downloadFile`/`downloadThumbnail` without `outPath`, and the
backup and metadata backup trees; it removes separators, control characters,
leading dots and Windows device names, and falls back to a name built from the
ID for an empty title. Originals-cache extensions are letters and digits only,
else `.bin`. A user-supplied path is used as is. `decryptFile` reads a missing
or non-string title as "" and rejects metadata that is not a JSON object.
- 2026-09-22: Rewrote the README API reference (and the Getting Started / usage
snippets) to match the shipped cache/API library on `next` (issue 53, issue
13). Documented `Library.open` and its options, the default-read vs `fresh()`
distinction (and that the CLI's read commands are fresh), the record types and
`snapshot()`/`subscribe()`, the `albums`/`photos`/`timeline` read surface,
`Photo` content methods, `thumbnails.ensure`, the `mldata` search surface,
`backup()`, the three request pools (10/5/25), and the on-disk cache layout;
noted the deferred content-hash integrity check (issue 68). Docs-only; no code
changed.
- 2026-09-22: Added resumable, deletion-aware enumeration to `Client` (issue 38,
closes issue 7). `collectionsSince`/`filesSince` take a starting cursor,
decrypt live records, surface tombstoned ids in a separate `deleted` list (a
@@ -335,6 +122,7 @@ declares one.
# Future Steps
- Tag v1.0.0.
- Future desktop client, separate repo:
- Electron app skeleton consuming this library.
- Local SQLite cache keyed on (collectionID, fileID, updationTime).
+336 -63
View File
@@ -1,79 +1,146 @@
#!/usr/bin/env node
import { input, password as passwordPrompt } from "@inquirer/prompts";
import { stdout, stderr } from "node:process";
import { input, password } from "@inquirer/prompts";
import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
import { join } from "node:path";
import { Command } from "commander";
import envPaths from "env-paths";
import { Client, type ClientSnapshot } from "../src/client.js";
import { init } from "../src/crypto/index.js";
import { runBackup } from "../src/backup.js";
import { runMetadataBackup } from "../src/metadata-backup.js";
import {
type CliContext,
loginCommand,
whoamiCommand,
logoutCommand,
collectionsCommand,
filesCommand,
getCommand,
getThumbCommand,
backupMetadataCommand,
backupCommand,
listMissingThumbnailsCommand,
fixMissingThumbnailsCommand,
} from "../src/cli-commands.js";
import { run as runCommand } from "../src/cli-run.js";
import { loadSession } from "../src/cli-session.js";
import { Client } from "../src/client.js";
import { VERSION } from "../src/index.js";
listMissingThumbnails,
fixMissingThumbnails,
} from "../src/thumbnails.js";
const paths = envPaths("quak", { suffix: "" });
const sessionPath = join(paths.data, "session.json");
const loadSession = (): ClientSnapshot | null => {
if (!existsSync(sessionPath)) return null;
try {
return JSON.parse(readFileSync(sessionPath, "utf-8")) as ClientSnapshot;
} catch {
return null;
}
};
const saveSession = (snapshot: ClientSnapshot): void => {
mkdirSync(paths.data, { recursive: true, mode: 0o700 });
writeFileSync(sessionPath, JSON.stringify(snapshot, null, 2), {
mode: 0o600,
});
};
const requireSession = (): Client => {
const snapshot = loadSession();
if (!snapshot) {
stderr.write(
`Not logged in. Run "quak login" first.\nSession file: ${sessionPath}\n`,
);
process.exit(1);
}
return Client.fromJSON(snapshot);
};
const prompt = async (message: string): Promise<string> => input({ message });
const promptSecret = async (message: string): Promise<string> =>
passwordPrompt({ message, mask: true });
const program = new Command();
program
.name("quak")
.description("CLI for the Ente end-to-end encrypted photo service")
.version(VERSION)
.option(
"--cache-dir <path>",
"Directory for the local metadata/content cache " +
"(default: the per-user cache directory)",
);
const context = (): CliContext => ({
stdout,
stderr,
sessionDir: paths.data,
cacheDir: program.opts<{ cacheDir?: string }>().cacheDir,
loadSession,
login: (opts) => Client.login(opts),
prompt: (message) => input({ message }),
promptSecret: (message) => password({ message, mask: true }),
});
const run = (command: Promise<number>): Promise<void> =>
runCommand(command, stdout, stderr, (code) => process.exit(code));
.version("0.0.0");
program
.command("login")
.description("Log in to an Ente account and save the session")
.action(() => run(loginCommand(context())));
.action(async () => {
await init();
const email = process.env.QUAK_EMAIL ?? (await prompt("Email"));
const password =
process.env.QUAK_PASSWORD ?? (await promptSecret("Password"));
stderr.write("Authenticating...\n");
try {
const client = await Client.login({
email,
password,
totp: async () => prompt("TOTP code: "),
emailOTP: async () => prompt("Email verification code: "),
});
saveSession(client.toJSON());
const info = client.whoami();
stderr.write(`Logged in as ${info.email} (user ${info.userID})\n`);
stderr.write(`Session saved to ${sessionPath}\n`);
} catch (err) {
stderr.write(
`Login failed: ${err instanceof Error ? err.message : err}\n`,
);
process.exit(1);
}
});
program
.command("whoami")
.description("Print the logged-in account")
.action(() => run(whoamiCommand(context())));
.action(() => {
const client = requireSession();
const info = client.whoami();
stdout.write(JSON.stringify(info) + "\n");
});
program
.command("logout")
.description("End the session on the server and delete the saved session")
.action(() => run(logoutCommand(context())));
.description("Delete the saved session")
.action(async () => {
if (existsSync(sessionPath)) {
const { unlinkSync } = await import("node:fs");
unlinkSync(sessionPath);
stderr.write("Session deleted.\n");
} else {
stderr.write("No session found.\n");
}
});
program
.command("collections")
.description("List all collections (albums)")
.option("--json", "Output as JSON array")
.action((opts: { json?: boolean }) =>
run(collectionsCommand(context(), opts)),
.action(async (opts: { json?: boolean }) => {
await init();
const client = requireSession();
const collections = await client.listCollections();
if (opts.json) {
stdout.write(
JSON.stringify(
collections.map((c) => ({
id: c.id,
name: c.name,
type: c.type,
ownerID: c.ownerID,
isShared: c.isShared,
updationTime: c.updationTime,
})),
null,
2,
) + "\n",
);
} else {
for (const c of collections) {
stdout.write(
`${c.id}\t${c.type}\t${c.name}${c.isShared ? " (shared)" : ""}\n`,
);
}
}
});
program
.command("files")
@@ -83,18 +150,93 @@ program
"Collection ID (from `quak collections`)",
)
.option("--json", "Output as JSON array")
.action((opts: { collection: string; json?: boolean }) =>
run(filesCommand(context(), opts)),
.action(async (opts: { collection: string; json?: boolean }) => {
await init();
const client = requireSession();
const collectionID = Number(opts.collection);
if (!Number.isFinite(collectionID)) {
stderr.write("Invalid collection ID\n");
process.exit(1);
}
const collections = await client.listCollections();
const col = collections.find((c) => c.id === collectionID);
if (!col) {
stderr.write(`Collection ${collectionID} not found\n`);
process.exit(1);
}
const files = await client.listFiles(col.id, col.key);
if (opts.json) {
stdout.write(
JSON.stringify(
files.map((f) => ({
id: f.id,
title: f.metadata.title,
fileType: f.metadata.fileType,
creationTime: f.metadata.creationTime,
collectionID: f.collectionID,
})),
null,
2,
) + "\n",
);
} else {
for (const f of files) {
stdout.write(
`${f.id}\t${f.metadata.fileType}\t${f.metadata.title}\n`,
);
}
}
});
program
.command("get")
.description("Download and decrypt a single file")
.argument("<fileID>", "File ID (from `quak files`)")
.option("--out <path>", "Output file path")
.option("--collection <id>", "Accepted for compatibility; ignored")
.action((fileID: string, opts: { out?: string }) =>
run(getCommand(context(), fileID, opts)),
.option(
"--collection <id>",
"Collection ID (required to look up the file key)",
)
.action(
async (
fileIDStr: string,
opts: { out?: string; collection?: string },
) => {
await init();
const client = requireSession();
const fileID = Number(fileIDStr);
if (!Number.isFinite(fileID)) {
stderr.write("Invalid file ID\n");
process.exit(1);
}
const collections = await client.listCollections();
let targetCol;
if (opts.collection) {
targetCol = collections.find(
(c) => c.id === Number(opts.collection),
);
}
// Search all collections (or the specified one) for the file
const searchCols = targetCol ? [targetCol] : collections;
for (const col of searchCols) {
const files = await client.listFiles(col.id, col.key);
const file = files.find((f) => f.id === fileID);
if (file) {
const result = await client.downloadFile(file, opts.out);
stderr.write(
`${result.bytesWritten} bytes -> ${result.path}\n`,
);
return;
}
}
stderr.write(`File ${fileID} not found\n`);
process.exit(1);
},
);
program
@@ -102,9 +244,49 @@ program
.description("Download and decrypt a thumbnail")
.argument("<fileID>", "File ID (from `quak files`)")
.option("--out <path>", "Output file path")
.option("--collection <id>", "Accepted for compatibility; ignored")
.action((fileID: string, opts: { out?: string }) =>
run(getThumbCommand(context(), fileID, opts)),
.option(
"--collection <id>",
"Collection ID (required to look up the file key)",
)
.action(
async (
fileIDStr: string,
opts: { out?: string; collection?: string },
) => {
await init();
const client = requireSession();
const fileID = Number(fileIDStr);
if (!Number.isFinite(fileID)) {
stderr.write("Invalid file ID\n");
process.exit(1);
}
const collections = await client.listCollections();
let targetCol;
if (opts.collection) {
targetCol = collections.find(
(c) => c.id === Number(opts.collection),
);
}
const searchCols = targetCol ? [targetCol] : collections;
for (const col of searchCols) {
const files = await client.listFiles(col.id, col.key);
const file = files.find((f) => f.id === fileID);
if (file) {
const result = await client.downloadThumbnail(
file,
opts.out,
);
stderr.write(
`${result.bytesWritten} bytes -> ${result.path}\n`,
);
return;
}
}
stderr.write(`File ${fileID} not found\n`);
process.exit(1);
},
);
program
@@ -118,9 +300,14 @@ program
"Download each file and extract full EXIF/IPTC/XMP metadata (slow)",
)
.option("--all", "Alias for --exif")
.action((dir: string, opts: { exif?: boolean; all?: boolean }) =>
run(backupMetadataCommand(context(), dir, opts)),
);
.action(async (dir: string, opts: { exif?: boolean; all?: boolean }) => {
await init();
const client = requireSession();
await runMetadataBackup(client, dir, {
exif: opts.exif || opts.all,
onProgress: (msg) => stderr.write(msg + "\n"),
});
});
program
.command("backup")
@@ -129,9 +316,35 @@ program
)
.argument("<dir>", "Output directory")
.option("--json", "Print result as JSON instead of human-readable summary")
.action((dir: string, opts: { json?: boolean }) =>
run(backupCommand(context(), dir, opts)),
.action(async (dir: string, opts: { json?: boolean }) => {
await init();
const client = requireSession();
stderr.write("Starting backup...\n");
const result = await runBackup(client, dir, (msg) => {
if (!opts.json) stderr.write(msg + "\n");
});
if (opts.json) {
stdout.write(JSON.stringify(result, null, 2) + "\n");
} else {
stderr.write("\n--- Backup complete ---\n");
stderr.write(` Total files: ${result.totalFiles}\n`);
stderr.write(` Downloaded: ${result.downloaded}\n`);
stderr.write(` Skipped: ${result.skipped}\n`);
stderr.write(` Failed: ${result.failed}\n`);
if (result.errors.length > 0) {
stderr.write("\nFailed files:\n");
for (const e of result.errors) {
stderr.write(
` [${e.collection}] ${e.title} (id ${e.fileID}): ${e.error}\n`,
);
}
}
}
process.exit(result.failed > 0 ? 1 : 0);
});
const helper = program
.command("helper")
@@ -141,9 +354,30 @@ helper
.command("list-missing-thumbnails")
.description("List files whose thumbnails are missing or empty")
.option("--json", "Output as JSON array")
.action((opts: { json?: boolean }) =>
run(listMissingThumbnailsCommand(context(), opts)),
.action(async (opts: { json?: boolean }) => {
await init();
const client = requireSession();
const missing = await listMissingThumbnails(client, (msg) => {
if (!opts.json) stderr.write(msg + "\n");
});
if (opts.json) {
stdout.write(JSON.stringify(missing, null, 2) + "\n");
} else {
if (missing.length === 0) {
stderr.write("No missing thumbnails found.\n");
} else {
stderr.write(
`\n${missing.length} file(s) with missing thumbnails:\n`,
);
for (const m of missing) {
stdout.write(
`${m.fileID}\t${m.title}\t${m.collection}\t${m.reason}\n`,
);
}
}
}
});
helper
.command("fix-missing-thumbnails")
@@ -155,9 +389,48 @@ helper
"Specific file IDs to fix (default: fix all missing)",
)
.option("--json", "Output as JSON")
.action((opts: { file?: string[]; json?: boolean }) =>
run(fixMissingThumbnailsCommand(context(), opts)),
);
.action(async (opts: { file?: string[]; json?: boolean }) => {
await init();
const client = requireSession();
let fileIDs: number[];
if (opts.file && opts.file.length > 0) {
fileIDs = opts.file.map(Number).filter(Number.isFinite);
} else {
stderr.write("Scanning for missing thumbnails...\n");
const missing = await listMissingThumbnails(client, (msg) => {
if (!opts.json) stderr.write(msg + "\n");
});
fileIDs = missing.map((m) => m.fileID);
if (fileIDs.length === 0) {
stderr.write("No missing thumbnails found.\n");
return;
}
stderr.write(`Found ${fileIDs.length} file(s) to fix.\n`);
}
const results = await fixMissingThumbnails(client, fileIDs, (msg) => {
if (!opts.json) stderr.write(msg + "\n");
});
if (opts.json) {
stdout.write(JSON.stringify(results, null, 2) + "\n");
} else {
const ok = results.filter((r) => r.success).length;
const fail = results.filter((r) => !r.success).length;
stderr.write(`\n--- Done ---\n`);
stderr.write(` Fixed: ${ok}\n`);
stderr.write(` Failed: ${fail}\n`);
if (fail > 0) {
stderr.write("\nFailed files:\n");
for (const r of results.filter((r) => !r.success)) {
stderr.write(` ${r.fileID}\t${r.title}\t${r.error}\n`);
}
}
}
process.exit(results.some((r) => !r.success) ? 1 : 0);
});
await init();
await program.parseAsync();
program.parse();
+6 -12
View File
@@ -9,23 +9,17 @@
"type": "git",
"url": "https://git.eeqj.de/sneak/quak.git"
},
"private": true,
"type": "module",
"engines": {
"node": ">=22"
},
"main": "./dist/src/index.js",
"types": "./dist/src/index.d.ts",
"exports": {
".": {
"types": "./dist/src/index.d.ts",
"import": "./dist/src/index.js"
},
"./package.json": "./package.json"
},
"bin": {
"quak": "./dist/bin/quak.js"
},
"files": [
"dist/",
"README.md",
"LICENSE"
],
"scripts": {
"build": "script/build",
"quak": "node ./dist/bin/quak.js",
@@ -35,6 +29,7 @@
},
"devDependencies": {
"@eslint/js": "9.38.0",
"@types/libsodium-wrappers-sumo": "0.8.2",
"@types/node": "22.18.13",
"eslint": "9.38.0",
"prettier": "3.8.1",
@@ -48,7 +43,6 @@
"env-paths": "4.0.0",
"exif-reader": "2.0.3",
"fast-srp-hap": "2.0.4",
"fflate": "0.8.3",
"jpeg-js": "0.4.4",
"libsodium-wrappers-sumo": "0.8.4"
}
-14
View File
@@ -45,24 +45,10 @@ for (const bin of bins) {
'
}
# src/index.ts imports ../package.json for the version, which tsc copies to
# dist/package.json. Running the built CLI proves that import resolves from
# dist/ and reports the version package.json declares.
verify_version() {
built="$(node dist/bin/quak.js --version)"
declared="$(node -p 'require("./package.json").version')"
if [ "$built" != "$declared" ]; then
echo "build: dist/bin/quak.js reports $built, package.json declares $declared" >&2
exit 1
fi
echo "build: dist/bin/quak.js reports version $built"
}
main() {
cd "$ROOT"
yarn run tsc
verify_entrypoints
verify_version
}
main "$@"
+13 -5
View File
@@ -1,11 +1,19 @@
#!/bin/sh
# script/check: run all checks (test, lint). Our own extension to
# scripts-to-rule-them-all. Both are Docker phases. Must not modify any
# files.
# scripts-to-rule-them-all. Must not modify any files.
#
# script/fmt-check is not called here, unlike the template: the lint
# phase already runs `prettier --check .`, so calling it would run
# prettier a second time over the same tree for the same verdict.
# The formatting check is part of lint, not a step of its own:
# script/lint builds Dockerfile.lint, which runs eslint AND
# `prettier --check .` as build steps. Calling script/fmt-check here as
# well would run prettier a second time over the same tree for the same
# verdict — the weaker of the two, since the host toolchain is whatever
# the working tree happens to have installed while the container's is
# digest-pinned. script/fmt-check remains a standalone entrypoint for
# asking the formatting question by itself.
#
# script/lint builds Dockerfile.lint, so this script requires docker and
# must never be run from inside a container: that is why the Dockerfile
# image runs script/test and script/build rather than this.
set -eu
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
+13 -17
View File
@@ -1,11 +1,15 @@
#!/bin/sh
# script/cibuild: run the CI build. The image's last stage depends on the
# lint and test phases, so this one build runs eslint, prettier and the
# suite once each and then compiles. Unlike the template it does not run
# script/check first, which would run lint and the tests a second time.
# --no-cache for the same reason as script/docker: the gate phases the
# final stage depends on are RUN steps, and a cached one is a check that
# did not run.
# script/cibuild: run the CI build, which is both images in a defined order.
#
# First script/lint, which builds Dockerfile.lint and is the one and only
# place linting happens — it goes first so a lint failure is reported before
# the slower suite runs. Then the Dockerfile image, which runs script/test
# and script/build. CHECK_EPOCH and LINT_EPOCH differ on every invocation, so
# neither the linters nor the suite can be served from Docker's cache: a
# green build here means the checks ran now, not that a previous run was
# remembered. The layers below the epochs (bootstrap, yarn install) are
# unaffected and stay cached. A build that omits the arguments fails by
# design.
set -eu
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
@@ -13,16 +17,8 @@ ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
main() {
cd "$ROOT"
# Own line: a failing command substitution inside an argument does
# not trip `set -e`, so the inline form degrades silently to an
# empty constant. VERSION is computed here because .dockerignore
# excludes .git, so `git describe` in a build stage yields an empty
# version without failing.
version="$(git describe --tags --always --dirty 2>/dev/null || true)"
[ -n "$version" ] || version="unknown"
docker build --no-cache \
--build-arg VERSION="$version" \
-t "$("$SCRIPT_DIR/projectname")" .
"$SCRIPT_DIR/lint"
docker build --build-arg CHECK_EPOCH="$(date +%s)" .
}
main "$@"
+5 -11
View File
@@ -1,8 +1,10 @@
#!/bin/sh
# script/docker: build the Docker image tagged with the project name.
# Identical in all repos; the tag comes from script/projectname.
# --no-cache because the gate phases the final stage depends on are RUN
# steps, and a cached one is a check that did not run.
# CHECK_EPOCH is passed for the same reason script/cibuild passes it: the
# Dockerfile refuses to build without it, so that no path to an image can
# quietly serve the test and build layers from cache. This builds the test
# and build image only; linting is a separate image, built by script/lint.
set -eu
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
@@ -10,15 +12,7 @@ ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
main() {
cd "$ROOT"
# Own line: a failing command substitution inside an argument does
# not trip `set -e`, so the inline form degrades silently to an
# empty constant. VERSION is computed here because .dockerignore
# excludes .git, so `git describe` in a build stage yields an empty
# version without failing.
version="$(git describe --tags --always --dirty 2>/dev/null || true)"
[ -n "$version" ] || version="unknown"
docker build --no-cache \
--build-arg VERSION="$version" \
docker build --build-arg CHECK_EPOCH="$(date +%s)" \
-t "$("$SCRIPT_DIR/projectname")" .
}
+14 -13
View File
@@ -1,23 +1,24 @@
#!/bin/sh
# script/lint: run the linter. Linting is a phase of the Dockerfile and
# this builds that phase alone; the linter is never installed or run on
# a developer host, where a shared result cache and a host-global lock
# make its answer untrustworthy.
# script/lint: run the linters. eslint and prettier are never run against
# the working tree from here: linting runs via docker only, one way,
# everywhere script/lint builds Dockerfile.lint, which COPYs the repo into
# the pinned node image and runs the linters as build steps. That works even
# when the docker daemon is remote and bind mounts are impossible.
#
# The phase is not the last stage in the file, so it is built only when
# --target names it. --no-cache because a cached lint layer is a lint
# that did not run. The tag makes each build replace the previous image
# instead of leaving a dangling one behind.
# LINT_EPOCH is passed on every invocation because no lint cache is wanted:
# on an unchanged tree Docker would otherwise serve the linter layers, having
# linted nothing, and still exit 0. Dockerfile.lint refuses to build without
# the argument, so no path to a lint result can quietly come from cache.
#
# Nothing that runs inside a container may call this script; see the header
# of Dockerfile.
set -eu
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
main() {
cd "$ROOT"
docker build --no-cache \
--target lint \
-t "$("$SCRIPT_DIR/projectname")-lint" .
docker build --build-arg LINT_EPOCH="$(date +%s)" -f Dockerfile.lint .
}
main "$@"
+11 -3
View File
@@ -4,9 +4,17 @@
#
# Runs lint but deliberately NOT the tests, so the TDD red-phase commit
# (failing tests, no implementation yet) can land. CI runs
# script/cibuild, whose image build includes the test phase, and so
# catches any branch that ships red. The lint phase includes the
# prettier check, so a badly formatted tree still fails the commit.
# script/cibuild, which builds both images and so catches any branch
# that ships red.
#
# The formatting check is still enforced here, because script/lint is a
# build of Dockerfile.lint and that runs `prettier --check .` as a build
# step: a badly formatted tree fails this hook, and therefore the
# commit. Calling script/fmt-check as well would only run prettier a
# second time over the same tree for the same verdict.
#
# script/lint is a docker build (Dockerfile.lint); docker is required to
# commit, which is the point of linting one way, everywhere.
set -eu
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
+16 -10
View File
@@ -1,19 +1,25 @@
#!/bin/sh
# script/test: run the test suite. Testing is a phase of the Dockerfile
# and this builds that phase alone, on the same terms as script/lint:
# --target because a phase that is not the last stage is built only when
# named, --no-cache because a cached test layer is a test that did not
# run, and a tag so each build replaces the previous image.
# script/test: run the test suite. Uses `timeout` (GNU coreutils) when
# available so the run is hard-capped at 30s; on macOS without
# coreutils the cap is skipped.
set -eu
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)"
ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)"
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
rerun_verbose() {
echo "--- Rerunning with verbose for details ---"
yarn run vitest run --reporter=verbose
exit 1
}
main() {
cd "$ROOT"
docker build --no-cache \
--target test \
-t "$("$SCRIPT_DIR/projectname")-test" .
TIMEOUT="$(command -v timeout 2>/dev/null || command -v gtimeout 2>/dev/null || true)"
if [ -n "$TIMEOUT" ]; then
"$TIMEOUT" 30s yarn run vitest run --reporter=dot || rerun_verbose
else
yarn run vitest run --reporter=dot || rerun_verbose
fi
}
main "$@"
+53 -103
View File
@@ -19,12 +19,14 @@ const DEFAULT_FILES_ORIGIN = "https://files.ente.io";
const DEFAULT_THUMBS_ORIGIN = "https://thumbnails.ente.io";
const CLIENT_PACKAGE = "berlin.sneak.quak";
// Two deadlines of different kinds. `requestTimeoutMs` bounds a whole JSON
// call. `downloadTimeoutMs` is an idle deadline: a file or thumbnail download
// is aborted only when no bytes have arrived for that long, so a large video on
// a slow link that keeps making progress is never cut off.
// Two deadlines rather than one, because a single number cannot serve both
// jobs. Thirty seconds is generous for a JSON call and short enough that a
// hung API connection cannot stall a backup for long. A file body is a
// different shape of problem: the deadline has to cover the whole transfer,
// which for a large video on a slow link is minutes, so a value sane for JSON
// would cancel legitimate downloads.
export const DEFAULT_REQUEST_TIMEOUT_MS = 30_000;
export const DEFAULT_DOWNLOAD_TIMEOUT_MS = 60_000;
export const DEFAULT_DOWNLOAD_TIMEOUT_MS = 600_000;
export interface ApiClientOptions {
apiOrigin?: string;
@@ -46,44 +48,18 @@ export interface StreamOptions {
retry?: boolean;
}
// An abort signal that fires once `ms` pass without a call to `restart`. It
// aborts with a `TimeoutError`, the same reason `AbortSignal.timeout()` gives,
// so the retry classifier treats an idle download exactly as it treats any
// other deadline. `stop` must be called when the download ends. The timer is
// unref'd, so even one left running never keeps the process alive.
const idleDeadline = (ms: number) => {
const controller = new AbortController();
let timer: ReturnType<typeof setTimeout> | undefined;
const stop = (): void => clearTimeout(timer);
const restart = (): void => {
stop();
timer = setTimeout(() => {
controller.abort(
new DOMException(
`download stalled: no bytes received for ${ms} ms`,
"TimeoutError",
),
);
}, ms);
timer.unref();
};
restart();
return { signal: controller.signal, restart, stop };
};
// Enforce the idle deadline over a response body, not merely over its headers.
// Enforce a deadline over a response body, not merely over its headers.
//
// `getFileStream` returns as soon as headers arrive; the bytes are pulled
// later, in the download layer. Whether the signal passed to `fetch` also
// tears down the body afterwards is up to the fetch implementation, so this
// wrapper makes it a property of quak instead: every read races the signal,
// each chunk that arrives restarts the deadline, and an abort errors the
// stream with the abort reason — which the retry classifier recognises.
// and an abort errors the stream with the abort reason — which the retry
// classifier recognises.
const deadlineStream = (
body: ReadableStream<Uint8Array>,
deadline: ReturnType<typeof idleDeadline>,
signal: AbortSignal,
): ReadableStream<Uint8Array> => {
const { signal } = deadline;
const reader = body.getReader();
let rejectOnAbort: (reason: unknown) => void = () => undefined;
const aborted = new Promise<never>((_resolve, reject) => {
@@ -97,10 +73,7 @@ const deadlineStream = (
const onAbort = (): void => rejectOnAbort(signal.reason);
if (signal.aborted) onAbort();
else signal.addEventListener("abort", onAbort, { once: true });
const release = (): void => {
deadline.stop();
signal.removeEventListener("abort", onAbort);
};
const release = (): void => signal.removeEventListener("abort", onAbort);
return new ReadableStream<Uint8Array>({
async pull(controller) {
@@ -111,7 +84,6 @@ const deadlineStream = (
controller.close();
return;
}
deadline.restart();
controller.enqueue(next.value);
} catch (err) {
release();
@@ -126,30 +98,6 @@ const deadlineStream = (
});
};
// The one place a request URL is built. `origin` may carry a base path (a
// self-hosted server behind a prefix) and may end in a slash; `path` may or
// may not start with one. Query parameters go only through `query`, which
// percent-encodes them: a `?` or `#` in `path` is an error, because
// `new URL` would otherwise quietly treat what follows as something else.
const buildURL = (
origin: string,
path: string,
query?: Record<string, string | number | undefined>,
): string => {
if (path.includes("?") || path.includes("#")) {
throw new Error(
`request path must not contain "?" or "#"; pass query parameters separately: ${path}`,
);
}
const url = new URL(
`${origin.replace(/\/+$/, "")}/${path.replace(/^\/+/, "")}`,
);
for (const [k, v] of Object.entries(query ?? {})) {
if (v !== undefined) url.searchParams.set(k, String(v));
}
return url.href;
};
export class ApiClient {
private readonly apiOrigin: string;
private readonly isCustomOrigin: boolean;
@@ -191,16 +139,11 @@ export class ApiClient {
this.token = undefined;
}
getAuthToken(): string | undefined {
return this.token;
}
// The policy this client was configured with, so that a caller wrapping a
// whole operation in its own `withRetry` — the download layer — runs under
// the same settings rather than under the library defaults.
// A copy, so the caller cannot change this client's settings through it.
getRetryOptions(): ResolvedRetryOptions {
return { ...this.retry };
return this.retry;
}
private headers(extra?: Record<string, string>): Record<string, string> {
@@ -254,10 +197,22 @@ export class ApiClient {
path: string,
query?: Record<string, string | number | undefined>,
): Promise<T> {
const url = buildURL(this.apiOrigin, path, query);
const url = new URL(path, this.apiOrigin + "/");
// new URL with a base resolves relative paths; ensure we keep the
// origin from apiOrigin even when path starts with /
url.protocol = new URL(this.apiOrigin).protocol;
url.host = new URL(this.apiOrigin).host;
url.pathname = path;
if (query) {
for (const [k, v] of Object.entries(query)) {
if (v !== undefined) {
url.searchParams.set(k, String(v));
}
}
}
// A GET changes nothing, so it is retried under the full policy.
return withRetry(async () => {
const resp = await this._fetch(url, {
const resp = await this._fetch(url.href, {
method: "GET",
headers: this.headers(),
signal: AbortSignal.timeout(this.requestTimeoutMs),
@@ -268,16 +223,16 @@ export class ApiClient {
}
async postJSON<T>(path: string, body: unknown): Promise<T> {
const url = buildURL(this.apiOrigin, path);
// Not idempotent: a POST is replayed only when `isSafeToReplay`
// says no request byte can have reached the server. The endpoints
// this covers are listed in the README under "Endpoints used".
//
// Redirects are not followed. The origin has already received the
// request when it answers with one, so a connection refused by the
// redirect target would look replay-safe when it is not. The API has
// no legitimate redirect, so one surfaces as an `ApiError` with its
// 3xx status, which is not retried.
const url = `${this.apiOrigin}${path}`;
// Idempotency: this reaches `/users/srp/create-session`,
// `/users/two-factor/verify` and `/users/ott`, all of which change
// server state — verifying a second factor consumes one of a small
// number of attempts. So a POST is replayed only on a failure that
// establishes no TCP connection to the server ever existed: DNS
// produced no address, or the peer refused the connection. A 5xx, a
// mid-flight reset, a routing errno (which Linux also delivers on an
// established socket) and a timeout are all left to the caller,
// because each of them can occur after the server has already acted.
return withRetry(
async () => {
const resp = await this._fetch(url, {
@@ -286,7 +241,6 @@ export class ApiClient {
"Content-Type": "application/json",
}),
body: JSON.stringify(body),
redirect: "manual",
signal: AbortSignal.timeout(this.requestTimeoutMs),
});
await this.throwIfError(resp);
@@ -301,8 +255,8 @@ export class ApiClient {
opts?: StreamOptions,
): Promise<ReadableStream<Uint8Array>> {
const url = this.isCustomOrigin
? buildURL(this.apiOrigin, `/files/download/${fileID}`)
: buildURL(this.filesOrigin, "/", { fileID });
? `${this.apiOrigin}/files/download/${fileID}`
: `${this.filesOrigin}/?fileID=${fileID}`;
return this.streamRequest(url, opts);
}
@@ -344,8 +298,10 @@ export class ApiClient {
}
async putJSON<T>(path: string, body: unknown): Promise<T> {
const url = buildURL(this.apiOrigin, path);
// Same replay and redirect rules as `postJSON`, for the same reasons.
const url = `${this.apiOrigin}${path}`;
// Same idempotency rule as `postJSON`, for the same reason: this
// reaches `/files/thumbnail`, which registers an uploaded thumbnail
// against a file.
return withRetry(
async () => {
const resp = await this._fetch(url, {
@@ -354,7 +310,6 @@ export class ApiClient {
"Content-Type": "application/json",
}),
body: JSON.stringify(body),
redirect: "manual",
signal: AbortSignal.timeout(this.requestTimeoutMs),
});
await this.throwIfError(resp);
@@ -380,8 +335,8 @@ export class ApiClient {
opts?: StreamOptions,
): Promise<ReadableStream<Uint8Array>> {
const url = this.isCustomOrigin
? buildURL(this.apiOrigin, `/files/preview/${fileID}`)
: buildURL(this.thumbsOrigin, "/", { fileID });
? `${this.apiOrigin}/files/preview/${fileID}`
: `${this.thumbsOrigin}/?fileID=${fileID}`;
return this.streamRequest(url, opts);
}
@@ -390,27 +345,22 @@ export class ApiClient {
opts?: StreamOptions,
): Promise<ReadableStream<Uint8Array>> {
const once = async (): Promise<ReadableStream<Uint8Array>> => {
// A fresh deadline per attempt. It also covers the wait for the
// headers, when no bytes have arrived either.
const deadline = idleDeadline(this.downloadTimeoutMs);
try {
// A fresh deadline per attempt, so a retry gets the whole budget
// rather than the remainder of the one that just expired.
const signal = AbortSignal.timeout(this.downloadTimeoutMs);
const resp = await this._fetch(url, {
method: "GET",
headers: this.headers(),
signal: deadline.signal,
signal,
});
await this.throwIfError(resp);
if (!resp.body) {
// Carries the status, and is not retryable: a response
// that arrived without a body is malformed, and asking
// again produces the same malformed response.
// Carries the status, and is not retryable: a response that
// arrived without a body is malformed, and asking again
// produces the same malformed response.
throw new ApiError("response body is null", resp.status);
}
return deadlineStream(resp.body, deadline);
} catch (err) {
deadline.stop();
throw err;
}
return deadlineStream(resp.body, signal);
};
return opts?.retry === false ? once() : withRetry(once, this.retry);
}
+103 -511
View File
@@ -1,72 +1,13 @@
// The backup command, rebuilt on the library API (issue #51).
//
// `lib.backup()` waits for a completed refresh of the library (a failed one
// fails the backup before any file is touched), then, for every file in scope,
// gets its original bytes onto disk under `downloadDirectory` and rebuilds the
// derived views (per-file sidecars, per-collection symlink trees,
// per-collection JSON) from the model. The on-disk layout is the historical
// one, unchanged:
//
// <downloadDirectory>/
// originals/<fileID>.<ext> the decrypted bytes
// originals/<fileID>.json per-file metadata sidecar
// collections/<name>/<title> symlink into ../../originals
// collections/<name>.json per-collection metadata
// failures.json durable ledger of unresolved failures
//
// Crash-safety rests on two properties. Bytes are present-means-complete: an
// original appears under `originals/` only via the content layer's atomic
// temp-then-rename, so a file that exists is whole and is never re-fetched — an
// interrupted run resumes by listing the directory. The derived views hold no
// unique state, so they are rebuilt every run; that repairs stale sidecars and
// missing or broken symlinks left by an earlier crash. A rebuild also removes
// the symlinks into originals/ that no longer belong to an album, and the
// directories of albums that no longer exist.
//
// Resilience (issue #8): no per-file condition aborts the run. A failed
// download or a failed symlink is caught, recorded in `failures.json` with a
// classification, a running attempt count, and the last-tried time, and the run
// continues. `result.failed` — and thus the CLI's exit code — stays non-zero
// while any failure remains unresolved and clears once every one succeeds. Each
// run reconciles the ledger against the files it attempted, so an entry for a
// file that has since left the library (deleted) or this run's scope is dropped
// rather than counted forever, which would poison a scheduled backup's exit code.
import {
lstatSync,
existsSync,
mkdirSync,
readdirSync,
readFileSync,
readlinkSync,
rmdirSync,
rmSync,
statSync,
symlinkSync,
writeFileSync,
} from "node:fs";
import { copyFile, rename, rm } from "node:fs/promises";
import { basename, dirname, extname, join, relative } from "node:path";
import { fsyncPath, removeLeftoverTempFiles } from "./download/index.js";
import { safeExtension, sanitizeFileName } from "./filename.js";
import type { Collection, EnteFile } from "./model/types.js";
export type ProgressCallback = (message: string) => void;
export interface BackupOptions {
// Where the backup tree lives. Required: with none, `backup()` throws
// before any network traffic. A library opened with a `downloadDirectory`
// supplies the default.
downloadDirectory?: string;
// Fetch and store full-resolution originals. Default true.
includeOriginals?: boolean;
// Also fetch and store thumbnails under `thumbnails/<fileID>.jpg`. Default
// false.
includeThumbnails?: boolean;
// Restrict the backup to albums with these names; others are left untouched.
onlyAlbumNames?: string[];
onProgress?: ProgressCallback;
}
import { join, relative, extname } from "node:path";
import type { Client } from "./client.js";
import type { EnteFile } from "./model/types.js";
export interface BackupError {
fileID: number;
@@ -76,488 +17,139 @@ export interface BackupError {
}
export interface BackupResult {
// Distinct files in scope this run.
totalFiles: number;
// Originals fetched (or copied from the cache) this run.
downloaded: number;
// Originals already present and left untouched.
skipped: number;
// Files with an unresolved failure after this run (the ledger size); the
// CLI exits non-zero while this is above zero. A file can be both
// downloaded and failed if its bytes landed but its symlink did not.
failed: number;
// This run's per-file errors, in encounter order.
errors: BackupError[];
}
// The slice of the library that backup drives. `Library` implements it; a test
// can drive backup with a stand-in.
export interface BackupLibrary {
refresh(): Promise<void>;
listCollections(): Collection[];
listFiles(collectionID: number): EnteFile[];
// Get an original's bytes onto disk through the content cache/pools,
// returning where they landed: `destination` when they were fetched now,
// otherwise wherever they already were (the cache, or a prior backup).
original(fileID: number, destination: string): Promise<{ path: string }>;
thumbnail(fileID: number): Promise<{ path: string }>;
}
export type ProgressCallback = (message: string) => void;
type FailureClass = "transient" | "permanent" | "unknown";
const sanitizePath = (name: string): string =>
name.replace(/[/\\:*?"<>|]/g, "_").replace(/^\.+/, "_");
interface FailureEntry {
fileID: number;
title: string;
classification: FailureClass;
attempts: number;
lastTriedAt: number;
error: string;
}
const LEDGER_VERSION = 1;
// The originals/ filename for a file: `<id><ext>`, the extension taken from the
// title (or `.bin`). Matches the content cache's own naming so a present check
// lines up with what a fetch would write.
const originalName = (file: EnteFile): string =>
`${file.id}${safeExtension(file.metadata.title)}`;
// A regular file with content is treated as complete. A zero-byte file is not:
// it is the shape an aborted write leaves and must be re-fetched.
const isPresent = (path: string): boolean => {
try {
const s = statSync(path);
return s.isFile() && s.size > 0;
} catch {
return false;
}
const originalFileName = (file: EnteFile): string => {
const ext = extname(file.metadata.title || "") || ".bin";
return `${file.id}${ext}`;
};
// Best-effort classification for the ledger. Retryable server/network problems
// are transient; refusals and local filesystem/decrypt errors are permanent;
// anything else is unknown. Both the error code and message are inspected.
const classify = (err: unknown): FailureClass => {
const e = err as NodeJS.ErrnoException;
const text =
`${e?.code ?? ""} ${err instanceof Error ? err.message : String(err)}`.toLowerCase();
if (
/timeout|timed out|econnreset|econnrefused|econnaborted|network|socket|eai_again|throttl|temporarily|429|500|502|503|504/.test(
text,
)
) {
return "transient";
}
if (
/enoent|eacces|eperm|eexist|eisdir|enotempty|erofs|enospc|not found|forbidden|unauthor|decrypt|truncat|401|403|404/.test(
text,
)
) {
return "permanent";
}
return "unknown";
};
export const runBackup = async (
client: Client,
outDir: string,
onProgress?: ProgressCallback,
): Promise<BackupResult> => {
const log = onProgress ?? (() => {});
const errorMessage = (err: unknown): string =>
err instanceof Error ? err.message : String(err);
mkdirSync(outDir, { recursive: true });
const originalsDir = join(outDir, "originals");
mkdirSync(originalsDir, { recursive: true });
const collectionsDir = join(outDir, "collections");
mkdirSync(collectionsDir, { recursive: true });
// Copy bytes into `dest` via a temp file in the same directory plus rename, so
// `dest` appears only once it is whole ("present means complete"). As in the
// download writer, the temp file is fsynced before the rename and the directory
// after it, so a power cut cannot leave a correctly named but short original.
// The temp name carries this process's ID so a later run can tell a leftover
// from a copy still in progress (see `removeLeftoverTempFiles`).
const copyAtomic = async (src: string, dest: string): Promise<void> => {
if (src === dest) return;
const tmp = join(
dirname(dest),
`.quak-backup-${basename(dest)}-${process.pid}-${Math.random()
.toString(36)
.slice(2)}.tmp`,
log("Fetching collections...");
const collections = await client.listCollections();
const downloadedIDs = new Set<number>();
let totalFiles = 0;
let downloaded = 0;
let skipped = 0;
let failed = 0;
const errors: BackupError[] = [];
for (const col of collections) {
const colDirName = sanitizePath(col.name || `collection-${col.id}`);
const colDir = join(collectionsDir, colDirName);
mkdirSync(colDir, { recursive: true });
log(`[${col.name}] Fetching file list...`);
const files = await client.listFiles(col.id, col.key);
log(`[${col.name}] ${files.length} file(s)`);
const collectionMeta: {
id: number;
name: string;
type: string;
files: { id: number; metadata: EnteFile["metadata"] }[];
} = {
id: col.id,
name: col.name,
type: col.type,
files: [],
};
for (const file of files) {
totalFiles++;
const origName = originalFileName(file);
const origPath = join(originalsDir, origName);
const linkName = sanitizePath(
file.metadata.title || `file-${file.id}`,
);
const linkPath = join(colDir, linkName);
if (!downloadedIDs.has(file.id)) {
if (existsSync(origPath) && statSync(origPath).size > 0) {
skipped++;
downloadedIDs.add(file.id);
} else {
try {
await copyFile(src, tmp);
await fsyncPath(tmp);
// `rename` replaces the destination's directory entry: an existing
// symlink at `dest` is replaced, not followed, and the new file has
// the temp file's permissions (copied from `src`).
await rename(tmp, dest);
await fsyncPath(dirname(dest));
} finally {
await rm(tmp, { force: true });
}
};
// Ensure `linkPath` is a symlink to `target`, rebuilding a missing, wrong, or
// non-symlink entry. Throws on failure (a directory in the way, no permission)
// so the caller records it and moves on rather than aborting the run.
const rebuildSymlink = (linkPath: string, target: string): void => {
try {
const st = lstatSync(linkPath);
if (st.isSymbolicLink() && readlinkSync(linkPath) === target) return;
} catch {
// Nothing there (or unreadable): fall through to create it.
}
// Remove a wrong symlink or stray file. `force` ignores a missing path but
// still refuses a directory (no `recursive`), which surfaces as a failure.
rmSync(linkPath, { force: true });
symlinkSync(target, linkPath);
};
// The on-disk names for the entries of one directory, keyed by ID. Each name
// is used as is unless another entry would get the same name, ignoring case
// (two names that differ only in case are one entry on a case-insensitive
// file system); then every entry sharing it gets ` (<id>)`, before the
// extension when `beforeExtension` is set. A name with an ID added can match
// another entry's own name (`IMG (6).JPG`), so this repeats until no name is
// shared. IDs are stable, so the names are too.
const namesByID = (
entries: { id: number; name: string }[],
beforeExtension: boolean,
): Map<number, string> => {
const withID = (id: number, name: string): string => {
const ext = beforeExtension ? extname(name) : "";
const stem = name.slice(0, name.length - ext.length);
return `${stem} (${id})${ext}`;
};
const names = new Map<number, string>();
for (const { id, name } of entries) names.set(id, name);
const suffixed = new Set<number>();
for (;;) {
const counts = new Map<string, number>();
for (const name of names.values()) {
const key = name.toLowerCase();
counts.set(key, (counts.get(key) ?? 0) + 1);
}
let changed = false;
for (const { id, name } of entries) {
if (suffixed.has(id)) continue;
if (counts.get(name.toLowerCase()) === 1) continue;
names.set(id, withID(id, name));
suffixed.add(id);
changed = true;
}
if (!changed) return names;
}
};
// Remove the symlinks in the album directory `dir` that point into
// `originalsDir` and are not named in `keep`. Nothing else in the directory
// is touched: anything else there was put there by the user.
const removeStaleLinks = (
dir: string,
keep: Set<string>,
originalsDir: string,
): void => {
const target = relative(dir, originalsDir);
for (const name of readdirSync(dir)) {
if (keep.has(name)) continue;
const path = join(dir, name);
if (
lstatSync(path).isSymbolicLink() &&
dirname(readlinkSync(path)) === target
) {
rmSync(path);
}
}
};
// Remove the directories under `collectionsDir` that an earlier run wrote for
// an album that is gone or renamed: a directory not named in `current` with a
// `<name>.json` beside it holding an album ID, which is what a run writes. Its
// symlinks into originals/ are removed; if that leaves it empty, it and its
// JSON are deleted, otherwise both stay for what the user put there.
const removeStaleAlbumDirs = (
collectionsDir: string,
current: Set<string>,
originalsDir: string,
): void => {
for (const entry of readdirSync(collectionsDir, { withFileTypes: true })) {
if (!entry.isDirectory() || current.has(entry.name)) continue;
const jsonPath = join(collectionsDir, `${entry.name}.json`);
try {
const album = JSON.parse(readFileSync(jsonPath, "utf-8")) as {
id?: unknown;
};
if (typeof album.id !== "number") continue;
} catch {
log(`[${col.name}] Downloading ${linkName}...`);
await client.downloadFile(file, origPath);
downloaded++;
downloadedIDs.add(file.id);
} catch (err) {
log(
`[${col.name}] FAILED ${linkName}: ${err instanceof Error ? err.message : err}`,
);
failed++;
errors.push({
fileID: file.id,
title: file.metadata.title,
collection: col.name,
error:
err instanceof Error
? err.message
: String(err),
});
continue;
}
const dir = join(collectionsDir, entry.name);
removeStaleLinks(dir, new Set(), originalsDir);
if (readdirSync(dir).length > 0) continue;
rmdirSync(dir);
rmSync(jsonPath);
}
};
}
const loadLedger = (path: string): Map<number, FailureEntry> => {
const ledger = new Map<number, FailureEntry>();
try {
const parsed = JSON.parse(readFileSync(path, "utf-8")) as {
files?: Record<string, FailureEntry>;
};
for (const entry of Object.values(parsed.files ?? {})) {
if (entry && typeof entry.fileID === "number") {
ledger.set(entry.fileID, entry);
}
}
} catch {
// No ledger yet, or an unreadable one: start clean.
}
return ledger;
};
const saveLedger = (path: string, ledger: Map<number, FailureEntry>): void => {
if (ledger.size === 0) {
rmSync(path, { force: true });
return;
}
const files: Record<string, FailureEntry> = {};
for (const [fileID, entry] of ledger) files[String(fileID)] = entry;
writeFileSync(
path,
JSON.stringify({ version: LEDGER_VERSION, files }, null, 2),
);
};
const writeSidecar = (path: string, file: EnteFile): void => {
const meta: Record<string, unknown> = {
// Write per-file metadata JSON alongside the original
const metaJsonPath = join(originalsDir, `${file.id}.json`);
if (!existsSync(metaJsonPath)) {
const fileMeta: Record<string, unknown> = {
id: file.id,
collectionID: file.collectionID,
ownerID: file.ownerID,
metadata: file.metadata,
};
if (file.magicMetadata) meta.magicMetadata = file.magicMetadata;
if (file.pubMagicMetadata) meta.pubMagicMetadata = file.pubMagicMetadata;
writeFileSync(path, JSON.stringify(meta, null, 2));
};
export const runBackup = async (
lib: BackupLibrary,
opts: BackupOptions,
): Promise<BackupResult> => {
const downloadDirectory = opts.downloadDirectory;
if (!downloadDirectory) {
throw new Error(
"backup requires a downloadDirectory (pass one to backup() or " +
"open the library with one)",
);
if (file.magicMetadata) {
fileMeta.magicMetadata = file.magicMetadata;
}
const includeOriginals = opts.includeOriginals ?? true;
const includeThumbnails = opts.includeThumbnails ?? false;
const log = opts.onProgress ?? (() => {});
const only = opts.onlyAlbumNames ? new Set(opts.onlyAlbumNames) : undefined;
log("Refreshing library...");
await lib.refresh();
const originalsDir = join(downloadDirectory, "originals");
const collectionsDir = join(downloadDirectory, "collections");
const thumbnailsDir = join(downloadDirectory, "thumbnails");
mkdirSync(originalsDir, { recursive: true });
mkdirSync(collectionsDir, { recursive: true });
if (includeThumbnails) mkdirSync(thumbnailsDir, { recursive: true });
removeLeftoverTempFiles(originalsDir);
removeLeftoverTempFiles(thumbnailsDir);
const ledgerPath = join(downloadDirectory, "failures.json");
const ledger = loadLedger(ledgerPath);
const now = Date.now();
// Collections in scope, and the distinct files across them (a file shared
// by two albums is one original).
const allCollections = lib.listCollections();
const collections = allCollections.filter((c) =>
only ? only.has(c.name) : true,
);
const collectionName = new Map<number, string>();
for (const c of collections) collectionName.set(c.id, c.name);
const distinct = new Map<number, EnteFile>();
const filesByCollection = new Map<number, EnteFile[]>();
for (const c of collections) {
const files = lib.listFiles(c.id);
filesByCollection.set(c.id, files);
for (const f of files) if (!distinct.has(f.id)) distinct.set(f.id, f);
if (file.pubMagicMetadata) {
fileMeta.pubMagicMetadata = file.pubMagicMetadata;
}
writeFileSync(metaJsonPath, JSON.stringify(fileMeta, null, 2));
}
const errors: BackupError[] = [];
const failedThisRun = new Set<number>();
let downloaded = 0;
let skipped = 0;
if (!existsSync(linkPath) && existsSync(origPath)) {
const target = relative(colDir, origPath);
symlinkSync(target, linkPath);
}
const recordFailure = (
file: EnteFile,
collection: string,
err: unknown,
): void => {
// Count at most one attempt per file per run: a file whose original
// and thumbnail both fail this run must not double its attempt count
// or appear twice in errors.
if (failedThisRun.has(file.id)) return;
const error = errorMessage(err);
errors.push({
fileID: file.id,
title: file.metadata.title,
collection,
error,
collectionMeta.files.push({
id: file.id,
metadata: file.metadata,
});
const prior = ledger.get(file.id);
ledger.set(file.id, {
fileID: file.id,
title: file.metadata.title,
classification: classify(err),
attempts: (prior?.attempts ?? 0) + 1,
lastTriedAt: now,
error,
});
failedThisRun.add(file.id);
};
// Phase 1: get the bytes. Fetch each pending original (and optional
// thumbnail) through the content cache/pools and place it under the backup
// tree; a present file is left as is.
if (includeOriginals) {
for (const [fileID, file] of distinct) {
const dest = join(originalsDir, originalName(file));
if (isPresent(dest)) {
skipped++;
continue;
}
try {
log(`Fetching original ${file.metadata.title} (${fileID})...`);
// A fetched original is written straight to `dest`; only one
// that was already cached elsewhere is copied.
const { path } = await lib.original(fileID, dest);
await copyAtomic(path, dest);
downloaded++;
} catch (err) {
log(
`FAILED original ${file.metadata.title}: ${errorMessage(err)}`,
);
recordFailure(
file,
collectionName.get(file.collectionID) ?? "",
err,
);
}
}
}
if (includeThumbnails) {
for (const [fileID, file] of distinct) {
const dest = join(thumbnailsDir, `${fileID}.jpg`);
if (isPresent(dest)) continue;
try {
const { path } = await lib.thumbnail(fileID);
await copyAtomic(path, dest);
} catch (err) {
recordFailure(
file,
collectionName.get(file.collectionID) ?? "",
err,
);
}
}
}
// Phase 2: rebuild the derived views from the model. Sidecars first, for
// every present original (this repairs stale ones).
if (includeOriginals) {
for (const [fileID, file] of distinct) {
const orig = join(originalsDir, originalName(file));
if (isPresent(orig)) {
writeSidecar(join(originalsDir, `${fileID}.json`), file);
}
}
}
// Then the per-collection symlink trees and JSON. Directory names are
// chosen across every album, not just those in scope, so a scoped run
// names an album the same as a full one and never takes the directory of
// an album it skipped. Stale entries are removed before anything is
// rebuilt, so on a case-insensitive file system removing an old name can
// never remove the new one.
const albumDirNames = namesByID(
allCollections.map((c) => ({
id: c.id,
name: sanitizeFileName(c.name, `collection-${c.id}`),
})),
false,
);
try {
removeStaleAlbumDirs(
collectionsDir,
new Set(albumDirNames.values()),
originalsDir,
);
} catch (err) {
log(`FAILED removing old album directories: ${errorMessage(err)}`);
}
for (const c of collections) {
const colDirName = albumDirNames.get(c.id)!;
const colDir = join(collectionsDir, colDirName);
mkdirSync(colDir, { recursive: true });
const files = filesByCollection.get(c.id) ?? [];
const linkNames = namesByID(
files.map((f) => ({
id: f.id,
name: sanitizeFileName(f.metadata.title, `file-${f.id}`),
})),
true,
);
try {
removeStaleLinks(colDir, new Set(linkNames.values()), originalsDir);
} catch (err) {
log(`FAILED removing old links in ${c.name}: ${errorMessage(err)}`);
}
const metaFiles: { id: number; metadata: EnteFile["metadata"] }[] = [];
for (const file of files) {
metaFiles.push({ id: file.id, metadata: file.metadata });
if (!includeOriginals) continue;
const orig = join(originalsDir, originalName(file));
if (!isPresent(orig)) continue;
const linkName = linkNames.get(file.id)!;
const linkPath = join(colDir, linkName);
try {
rebuildSymlink(linkPath, relative(colDir, orig));
} catch (err) {
log(
`FAILED symlink ${c.name}/${linkName}: ${errorMessage(err)}`,
);
recordFailure(file, c.name, err);
}
}
writeFileSync(
join(collectionsDir, `${colDirName}.json`),
JSON.stringify(
{ id: c.id, name: c.name, type: c.type, files: metaFiles },
null,
2,
),
JSON.stringify(collectionMeta, null, 2),
);
}
// Reconcile the ledger against what this run actually attempted: an entry
// survives only for a file that failed this run. A file that succeeded had
// its failure resolved; a file gone from the library (deleted) or outside
// this run's scope is not something this run can resolve, so keeping its
// stale entry would keep the exit code non-zero forever — a single
// since-deleted photo would fail every future scheduled backup.
for (const fileID of [...ledger.keys()]) {
if (!failedThisRun.has(fileID)) ledger.delete(fileID);
}
saveLedger(ledgerPath, ledger);
return {
totalFiles: distinct.size,
downloaded,
skipped,
failed: ledger.size,
errors,
};
return { totalFiles, downloaded, skipped, failed, errors };
};
-536
View File
@@ -1,536 +0,0 @@
// The CLI's commands as plain functions.
//
// Each command takes its options and a `CliContext` and resolves to the exit
// code; a thrown error is left to the caller. Nothing here calls
// `process.exit`: `bin/quak.ts` wires these to the command line, and `run` in
// `cli-run.ts` prints a thrown error as one line and exits once output has
// drained. Output must stay byte-identical (see `cli-output.ts`).
import {
copyFileSync,
existsSync,
mkdirSync,
unlinkSync,
writeFileSync,
} from "node:fs";
import { join } from "node:path";
import {
type Client,
type ClientSnapshot,
type LoginOptions,
} from "./client.js";
import { init } from "./crypto/index.js";
import {
defaultCacheDirectory,
Library,
type LibraryClient,
} from "./library/index.js";
import {
fileListRow,
fileListLine,
originalName,
thumbnailName,
} from "./cli-output.js";
import { freshCollections, freshFiles, freshFile } from "./cli-read.js";
import { runMetadataBackup } from "./metadata-backup.js";
import { listMissingThumbnails, fixMissingThumbnails } from "./thumbnails.js";
export interface CliContext {
stdout: { write(text: string): unknown };
stderr: { write(text: string): unknown };
// Directory holding `session.json`.
sessionDir: string;
// The `--cache-dir` global, or undefined to let the library pick its
// per-user default keyed by the account id.
cacheDir?: string;
// Reads the session file into a client, or null when there is none. The
// CLI passes `loadSession` from `cli-session.ts`; tests pass a fake client.
loadSession: (path: string) => Client | null;
// Used by `login` only. The CLI passes `Client.login` and terminal
// prompts; tests pass fakes.
login: (opts: LoginOptions) => Promise<Client>;
prompt: (message: string) => Promise<string>;
// Like `prompt`, but the answer is masked as it is typed.
promptSecret: (message: string) => Promise<string>;
}
const sessionPath = (ctx: CliContext): string =>
join(ctx.sessionDir, "session.json");
// Write the session readable by its owner only, in a directory only its owner
// can enter.
export const saveSession = (
sessionDir: string,
snapshot: ClientSnapshot,
): void => {
mkdirSync(sessionDir, { recursive: true, mode: 0o700 });
writeFileSync(
join(sessionDir, "session.json"),
JSON.stringify(snapshot, null, 2),
{ mode: 0o600 },
);
};
// The saved client, or undefined after telling the user why there is none.
const requireSession = (ctx: CliContext): Client | undefined => {
let client: Client | null;
try {
client = ctx.loadSession(sessionPath(ctx));
} catch (err) {
ctx.stderr.write(
`${err instanceof Error ? err.message : err}\n` +
`Run "quak logout" and then "quak login" to replace it.\n`,
);
return undefined;
}
if (!client) {
ctx.stderr.write(
`Not logged in. Run "quak login" first.\nSession file: ${sessionPath(ctx)}\n`,
);
return undefined;
}
return client;
};
// A library client that omits `fetchMLData`, so the point commands below do not
// kick the library's background ML backfill: they read metadata, or fetch one
// file's content, and exit. `backup` and `backup-metadata` handle ML on their
// own terms. The content source is kept so `get`/`get-thumb`/`--exif` can fetch
// originals through the on-disk cache.
const readLibraryClient = (client: Client): LibraryClient => ({
whoami: () => client.whoami(),
collectionsSince: (args) => client.collectionsSince(args),
filesSince: (args) => client.filesSince(args),
contentSource: () => client.contentSource(),
});
// Open a library for a single point command: the aggressive background precache
// (issue #48) is off — a one-shot `collections` or `get` must not start
// downloading the whole account — and the refresh interval is long so no second
// refresh fires mid-command.
const openReadLibrary = (ctx: CliContext, client: Client): Promise<Library> =>
Library.open({
client: readLibraryClient(client),
cacheDirectory: ctx.cacheDir,
refreshIntervalSeconds: 3600,
precacheThumbnails: false,
precacheOriginals: false,
});
export const loginCommand = async (ctx: CliContext): Promise<number> => {
await init();
const email = process.env.QUAK_EMAIL ?? (await ctx.prompt("Email"));
const password =
process.env.QUAK_PASSWORD ?? (await ctx.promptSecret("Password"));
ctx.stderr.write("Authenticating...\n");
try {
const client = await ctx.login({
email,
password,
totp: async () => ctx.prompt("TOTP code: "),
emailOTP: async () => ctx.prompt("Email verification code: "),
});
saveSession(ctx.sessionDir, client.toJSON());
const info = client.whoami();
ctx.stderr.write(`Logged in as ${info.email} (user ${info.userID})\n`);
ctx.stderr.write(`Session saved to ${sessionPath(ctx)}\n`);
} catch (err) {
ctx.stderr.write(
`Login failed: ${err instanceof Error ? err.message : err}\n`,
);
return 1;
}
return 0;
};
export const whoamiCommand = async (ctx: CliContext): Promise<number> => {
await init();
const client = requireSession(ctx);
if (!client) return 1;
const info = client.whoami();
ctx.stdout.write(JSON.stringify(info) + "\n");
return 0;
};
// Ends the session on the server, then deletes the session file even when that
// failed, and exits 1 if it did. The cache is left in place; the user is told
// where it is.
export const logoutCommand = async (ctx: CliContext): Promise<number> => {
const path = sessionPath(ctx);
if (!existsSync(path)) {
ctx.stderr.write("No session found.\n");
return 0;
}
await init();
let cacheDir = ctx.cacheDir;
let failure: string | undefined;
try {
const client = ctx.loadSession(path);
if (client) {
cacheDir ??= defaultCacheDirectory(client.whoami().userID);
await client.logoutOnServer();
client.logout();
}
} catch (err) {
failure = err instanceof Error ? err.message : String(err);
}
unlinkSync(path);
if (failure === undefined) {
ctx.stderr.write("Session ended on the server.\n");
} else {
ctx.stderr.write(
`Could not end the session on the server: ${failure}\n`,
);
}
ctx.stderr.write("Session deleted.\n");
if (cacheDir !== undefined) {
ctx.stderr.write(
`Cache directory ${cacheDir} still holds decrypted data; delete it to remove that data.\n`,
);
}
return failure === undefined ? 0 : 1;
};
export const collectionsCommand = async (
ctx: CliContext,
opts: { json?: boolean },
): Promise<number> => {
await init();
const client = requireSession(ctx);
if (!client) return 1;
const lib = await openReadLibrary(ctx, client);
try {
// Force a server round-trip and list in enumeration order (issue #36
// amendment, issue #52): the pre-library CLI printed current state in
// this order, not the albums projection's newest-first order.
const collections = await freshCollections(lib);
if (opts.json) {
ctx.stdout.write(
JSON.stringify(
collections.map((c) => ({
id: c.id,
name: c.name,
type: c.type,
ownerID: c.ownerID,
isShared: c.isShared,
updationTime: c.updationTime,
})),
null,
2,
) + "\n",
);
} else {
for (const c of collections) {
ctx.stdout.write(
`${c.id}\t${c.type}\t${c.name}${c.isShared ? " (shared)" : ""}\n`,
);
}
}
return 0;
} finally {
await lib.close();
}
};
export const filesCommand = async (
ctx: CliContext,
opts: { collection: string; json?: boolean },
): Promise<number> => {
await init();
const client = requireSession(ctx);
if (!client) return 1;
const collectionID = Number(opts.collection);
if (!Number.isFinite(collectionID)) {
ctx.stderr.write("Invalid collection ID\n");
return 1;
}
const lib = await openReadLibrary(ctx, client);
try {
// Force a server round-trip and list in enumeration order (issue #36
// amendment, issue #52). Each file prints from its own decrypted
// metadata (raw title, microsecond creationTime) via cli-output, and in
// the pre-library CLI's enumeration order, not the projection's
// newest-first order.
const files = await freshFiles(lib, collectionID);
if (!files) {
ctx.stderr.write(`Collection ${collectionID} not found\n`);
return 1;
}
if (opts.json) {
ctx.stdout.write(
JSON.stringify(files.map(fileListRow), null, 2) + "\n",
);
} else {
for (const file of files) {
ctx.stdout.write(fileListLine(file) + "\n");
}
}
return 0;
} finally {
await lib.close();
}
};
export const getCommand = async (
ctx: CliContext,
fileIDStr: string,
opts: { out?: string },
): Promise<number> => {
await init();
const client = requireSession(ctx);
if (!client) return 1;
const fileID = Number(fileIDStr);
if (!Number.isFinite(fileID)) {
ctx.stderr.write("Invalid file ID\n");
return 1;
}
const lib = await openReadLibrary(ctx, client);
try {
// Force a server round-trip so the file resolves against current state
// (issue #36 amendment, issue #52).
const resolved = await freshFile(lib, fileID);
if (!resolved) {
ctx.stderr.write(`File ${fileID} not found\n`);
return 1;
}
const { photo, file } = resolved;
const result = await photo.original();
// Default name is the file's own title, as the pre-library CLI used
// (not the editedName-preferring projection title) (issue #52).
const outPath = opts.out ?? originalName(file);
copyFileSync(result.path, outPath);
ctx.stderr.write(`${result.bytes} bytes -> ${outPath}\n`);
return 0;
} finally {
await lib.close();
}
};
export const getThumbCommand = async (
ctx: CliContext,
fileIDStr: string,
opts: { out?: string },
): Promise<number> => {
await init();
const client = requireSession(ctx);
if (!client) return 1;
const fileID = Number(fileIDStr);
if (!Number.isFinite(fileID)) {
ctx.stderr.write("Invalid file ID\n");
return 1;
}
const lib = await openReadLibrary(ctx, client);
try {
// Force a server round-trip so the file resolves against current state
// (issue #36 amendment, issue #52).
const resolved = await freshFile(lib, fileID);
if (!resolved) {
ctx.stderr.write(`File ${fileID} not found\n`);
return 1;
}
const { photo, file } = resolved;
const result = await photo.thumbnail();
// Default name is thumb_<file's own title>, as the pre-library CLI
// used (not the projection title) (issue #52).
const outPath = opts.out ?? thumbnailName(file);
copyFileSync(result.path, outPath);
ctx.stderr.write(`${result.bytes} bytes -> ${outPath}\n`);
return 0;
} finally {
await lib.close();
}
};
export const backupMetadataCommand = async (
ctx: CliContext,
dir: string,
opts: { exif?: boolean; all?: boolean },
): Promise<number> => {
await init();
const client = requireSession(ctx);
if (!client) return 1;
const lib = await openReadLibrary(ctx, client);
try {
// Refresh first so the dump holds current account state, not what the
// cache last held; a failed refresh throws.
await lib.fresh();
const { failedMLBatches } = await runMetadataBackup(lib, client, dir, {
exif: opts.exif || opts.all,
onProgress: (msg) => ctx.stderr.write(msg + "\n"),
});
return failedMLBatches > 0 ? 1 : 0;
} finally {
await lib.close();
}
};
export const backupCommand = async (
ctx: CliContext,
dir: string,
opts: { json?: boolean },
): Promise<number> => {
await init();
const client = requireSession(ctx);
if (!client) return 1;
ctx.stderr.write("Starting backup...\n");
// The precache is off: the backup fetches what it needs, and must not
// also fill the cache with every thumbnail and the recent originals.
const lib = await Library.open({
client,
downloadDirectory: dir,
cacheDirectory: ctx.cacheDir,
precacheThumbnails: false,
precacheOriginals: false,
});
try {
const result = await lib.backup({
downloadDirectory: dir,
onProgress: (msg) => {
if (!opts.json) ctx.stderr.write(msg + "\n");
},
});
if (opts.json) {
ctx.stdout.write(JSON.stringify(result, null, 2) + "\n");
} else {
ctx.stderr.write("\n--- Backup complete ---\n");
ctx.stderr.write(` Total files: ${result.totalFiles}\n`);
ctx.stderr.write(` Downloaded: ${result.downloaded}\n`);
ctx.stderr.write(` Skipped: ${result.skipped}\n`);
ctx.stderr.write(` Failed: ${result.failed}\n`);
if (result.errors.length > 0) {
ctx.stderr.write("\nFailed files:\n");
for (const e of result.errors) {
ctx.stderr.write(
` [${e.collection}] ${e.title} (id ${e.fileID}): ${e.error}\n`,
);
}
}
}
return result.failed > 0 ? 1 : 0;
} finally {
await lib.close();
}
};
export const listMissingThumbnailsCommand = async (
ctx: CliContext,
opts: { json?: boolean },
): Promise<number> => {
await init();
const client = requireSession(ctx);
if (!client) return 1;
const lib = await openReadLibrary(ctx, client);
try {
// Refresh first so files added since the cache was written are
// checked; a failed refresh throws.
await lib.fresh();
const missing = await listMissingThumbnails(lib, client, (msg) => {
if (!opts.json) ctx.stderr.write(msg + "\n");
});
if (opts.json) {
ctx.stdout.write(JSON.stringify(missing, null, 2) + "\n");
} else {
if (missing.length === 0) {
ctx.stderr.write("No missing thumbnails found.\n");
} else {
ctx.stderr.write(
`\n${missing.length} file(s) with missing thumbnails:\n`,
);
for (const m of missing) {
ctx.stdout.write(
`${m.fileID}\t${m.title}\t${m.collection}\t${m.reason}\n`,
);
}
}
}
return 0;
} finally {
await lib.close();
}
};
export const fixMissingThumbnailsCommand = async (
ctx: CliContext,
opts: { file?: string[]; json?: boolean },
): Promise<number> => {
await init();
const client = requireSession(ctx);
if (!client) return 1;
const lib = await openReadLibrary(ctx, client);
try {
// Refresh first so files added since the cache was written are found;
// a failed refresh throws.
await lib.fresh();
let fileIDs: number[];
if (opts.file && opts.file.length > 0) {
fileIDs = opts.file.map(Number).filter(Number.isFinite);
} else {
ctx.stderr.write("Scanning for missing thumbnails...\n");
const missing = await listMissingThumbnails(lib, client, (msg) => {
if (!opts.json) ctx.stderr.write(msg + "\n");
});
fileIDs = missing.map((m) => m.fileID);
if (fileIDs.length === 0) {
ctx.stderr.write("No missing thumbnails found.\n");
return 0;
}
ctx.stderr.write(`Found ${fileIDs.length} file(s) to fix.\n`);
}
const results = await fixMissingThumbnails(
lib,
client,
fileIDs,
(msg) => {
if (!opts.json) ctx.stderr.write(msg + "\n");
},
);
if (opts.json) {
ctx.stdout.write(JSON.stringify(results, null, 2) + "\n");
} else {
const fixed = results.filter((r) => r.status === "fixed").length;
const skipped = results.filter(
(r) => r.status === "skipped",
).length;
const failed = results.filter((r) => r.status === "failed").length;
ctx.stderr.write(`\n--- Done ---\n`);
ctx.stderr.write(` Fixed: ${fixed}\n`);
ctx.stderr.write(` Skipped: ${skipped}\n`);
ctx.stderr.write(` Failed: ${failed}\n`);
if (skipped > 0) {
ctx.stderr.write("\nSkipped:\n");
for (const r of results.filter((r) => r.status === "skipped")) {
ctx.stderr.write(
` ${r.fileID}\t${r.title}\t${r.reason}\n`,
);
}
}
if (failed > 0) {
ctx.stderr.write("\nFailed files:\n");
for (const r of results.filter((r) => r.status === "failed")) {
ctx.stderr.write(
` ${r.fileID}\t${r.title}\t${r.reason}\n`,
);
}
}
}
return results.some((r) => r.status === "failed") ? 1 : 0;
} finally {
await lib.close();
}
};
-43
View File
@@ -1,43 +0,0 @@
// How the CLI presents a file's identity in `files`, `get`, and `get-thumb`.
//
// These read the file's own decrypted metadata — the raw title and the
// creationTime in microseconds — rather than the `PhotoRecord` projection the
// rest of the library exposes. The projection prefers `editedName`/`editedTime`
// and reports time in milliseconds, which is right for a photo browser but
// would change the CLI's externally-visible output. The pre-library CLI printed
// `metadata.title` and `metadata.creationTime` and named downloads after
// `metadata.title`, and issue #52 requires that output stay byte-identical, so
// the commands shape their output from the raw `EnteFile` through here.
import { sanitizeFileName } from "./filename.js";
import type { EnteFile, FileType, Microseconds } from "./model/types.js";
// One row of `quak files --json`.
export interface FileListRow {
id: number;
title: string;
fileType: FileType;
creationTime: Microseconds;
collectionID: number;
}
export const fileListRow = (file: EnteFile): FileListRow => ({
id: file.id,
title: file.metadata.title,
fileType: file.metadata.fileType,
creationTime: file.metadata.creationTime,
collectionID: file.collectionID,
});
// One line of `quak files` in its human, tab-separated form.
export const fileListLine = (file: EnteFile): string =>
`${file.id}\t${file.metadata.fileType}\t${file.metadata.title}`;
// Default output path for `quak get` when `--out` is not given. The title comes
// from the server, so it is sanitized; `--out` is the user's and is used as is.
export const originalName = (file: EnteFile): string =>
sanitizeFileName(file.metadata.title, `file-${file.id}`);
// Default output path for `quak get-thumb` when `--out` is not given.
export const thumbnailName = (file: EnteFile): string =>
`thumb_${originalName(file)}`;
-62
View File
@@ -1,62 +0,0 @@
// How the CLI's read commands obtain current data.
//
// `collections`, `files --collection`, `get`, and `get-thumb` must answer for
// the account's state at the moment the command runs, not for whatever the
// local cache last happened to hold (owner amendment, issue #36). Each helper
// therefore forces a server round-trip through `Library.fresh()` and only then
// reads — so a collection, file, or metadata change made elsewhere is visible.
//
// `collections` and `files` also list in the library's own enumeration order —
// `listCollections()`/`listFiles()`, the order the pre-library CLI printed —
// rather than the `albums`/`photos` projection's newest-first order, which
// re-sorts the rows. The field values still come from each record's raw
// metadata via `cli-output.ts`.
import type { Collection, EnteFile } from "./model/types.js";
import type { Photo, PhotosAPI } from "./library/index.js";
// The slice of `Library` these helpers read. `Library` satisfies it
// structurally; a test can drive them with a stand-in that records the
// `fresh()` call and serves records in a known enumeration order.
export interface FreshReadLibrary {
fresh(): Promise<unknown>;
listCollections(): Collection[];
getCollection(id: number): Collection | undefined;
listFiles(collectionID: number): EnteFile[];
getFileByID(fileID: number): EnteFile | undefined;
photos: Pick<PhotosAPI, "byID">;
}
// Every live collection, current as of a forced refresh, in enumeration order.
export const freshCollections = async (
lib: FreshReadLibrary,
): Promise<Collection[]> => {
await lib.fresh();
return lib.listCollections();
};
// The files of one collection, current as of a forced refresh, in enumeration
// order. `undefined` (not an empty list) when the collection does not exist, so
// the caller can tell "no such collection" from "an empty collection".
export const freshFiles = async (
lib: FreshReadLibrary,
collectionID: number,
): Promise<EnteFile[] | undefined> => {
await lib.fresh();
if (!lib.getCollection(collectionID)) return undefined;
return lib.listFiles(collectionID);
};
// One file, current as of a forced refresh, resolved to both its content
// handle (`Photo`, for fetching bytes) and its raw record (`EnteFile`, for the
// default output name and field values). `undefined` when the file is unknown.
export const freshFile = async (
lib: FreshReadLibrary,
fileID: number,
): Promise<{ photo: Photo; file: EnteFile } | undefined> => {
await lib.fresh();
const photo = lib.photos.byID({ fileID });
const file = lib.getFileByID(fileID);
if (!photo || !file) return undefined;
return { photo, file };
};
-36
View File
@@ -1,36 +0,0 @@
// Runs one CLI command for `bin/quak.ts` and exits with its code.
import type { Writable } from "node:stream";
// Run a command and exit with its code once stdout/stderr have drained.
// Exiting before the drain can truncate piped output, and the library can keep
// the event loop alive after a command returns, so a plain return could hang.
// An error the command throws is printed as one `quak: MESSAGE` line, without
// the stack trace, and exits 1.
export const run = async (
command: Promise<number>,
stdout: Writable,
stderr: Writable,
exit: (code: number) => void,
): Promise<void> => {
let code: number;
try {
code = await command;
} catch (err) {
stderr.write(
`quak: ${err instanceof Error ? err.message : String(err)}\n`,
);
code = 1;
}
const pending = [stdout, stderr].filter((s) => s.writableLength > 0);
if (pending.length === 0) {
exit(code);
return;
}
let remaining = pending.length;
for (const s of pending) {
s.once("drain", () => {
if (--remaining === 0) exit(code);
});
}
};
-26
View File
@@ -1,26 +0,0 @@
// How the CLI reads its saved session file back into a `Client`.
//
// A missing file means "not logged in" and returns null. A file that exists but
// cannot be read back into a client (bad JSON, a missing field, a key of the
// wrong length) throws an error saying the session file is corrupt, so the CLI
// can tell the user which of the two it is. Needs `init()` first.
import { existsSync, readFileSync } from "node:fs";
import type { ApiClientOptions } from "./api/client.js";
import { Client } from "./client.js";
export const loadSession = (
path: string,
apiOptions?: ApiClientOptions,
): Client | null => {
if (!existsSync(path)) return null;
try {
return Client.fromJSON(
JSON.parse(readFileSync(path, "utf-8")),
apiOptions,
);
} catch (err) {
const reason = err instanceof Error ? err.message : String(err);
throw new Error(`Session file ${path} is corrupt: ${reason}`);
}
};
+11 -70
View File
@@ -125,57 +125,18 @@ export class Client {
);
}
// Restore a client from a `toJSON()` snapshot. The snapshot usually comes
// straight from `JSON.parse` of a file on disk, so every field is checked
// before use; a bad one throws an error naming it. Needs `init()` first.
static fromJSON(snapshot: unknown, apiOptions?: ApiClientOptions): Client {
const invalid = (field: string, problem: string): Error =>
new Error(`Invalid session data: ${field} ${problem}`);
if (typeof snapshot !== "object" || snapshot === null) {
throw new Error("Invalid session data: not a JSON object");
}
const s = snapshot as Record<string, unknown>;
for (const field of ["email", "token"]) {
if (typeof s[field] !== "string" || s[field] === "") {
throw invalid(field, "must be a non-empty string");
}
}
if (!Number.isInteger(s.userID)) {
throw invalid("userID", "must be an integer");
}
const key = (field: string): Uint8Array => {
const value = s[field];
if (typeof value !== "string") {
throw invalid(field, "must be a base64 string");
}
let bytes: Uint8Array;
try {
bytes = fromBase64(value);
} catch {
throw invalid(field, "is not valid base64");
}
// The master key (secretbox) and the key pair (box) are all 32 bytes.
if (bytes.length !== 32) {
throw invalid(
field,
`must decode to 32 bytes, got ${bytes.length}`,
);
}
return bytes;
};
const api = new ApiClient({
...apiOptions,
authToken: s.token as string,
});
static fromJSON(
snapshot: ClientSnapshot,
apiOptions?: ApiClientOptions,
): Client {
const api = new ApiClient({ ...apiOptions, authToken: snapshot.token });
return new Client(
api,
s.email as string,
s.userID as number,
key("masterKey"),
key("secretKey"),
key("publicKey"),
snapshot.email,
snapshot.userID,
fromBase64(snapshot.masterKey),
fromBase64(snapshot.secretKey),
fromBase64(snapshot.publicKey),
);
}
@@ -203,37 +164,19 @@ export class Client {
toJSON(): ClientSnapshot {
this.assertLoggedIn();
const token = this.api.getAuthToken();
if (!token) {
throw new Error("Cannot serialize client: it has no auth token");
}
return {
email: this.email,
userID: this.userID,
token,
token: this.api["token"]!,
masterKey: toBase64(this.masterKey),
secretKey: toBase64(this.secretKey),
publicKey: toBase64(this.publicKey),
};
}
// Ends this client's session on the server (`POST /users/logout`), so the
// token stops working everywhere, including in any saved copy of it. This
// client is left as it was; call `logout()` to clear it.
async logoutOnServer(): Promise<void> {
this.assertLoggedIn();
await this.api.postJSON("/users/logout", {});
}
// Zeroes the key buffers in place, so any copy of the reference held
// elsewhere is wiped too. Every method checks `assertLoggedIn` before
// touching the keys, so nothing decrypts with the zeroed keys.
logout(): void {
this.loggedOut = true;
this.api.clearAuthToken();
this.masterKey.fill(0);
this.secretKey.fill(0);
this.publicKey.fill(0);
}
// Enumerate collections changed since `sinceTime`. Live collections are
@@ -249,8 +192,6 @@ export class Client {
const { collections: raws } = await this.api.getJSON<{
collections: RawCollection[];
}>("/collections/v2", { sinceTime: args.sinceTime });
// logout() may have zeroed the keys while the request was in flight.
this.assertLoggedIn();
const collections: Collection[] = [];
const deleted: number[] = [];
-22
View File
@@ -1,22 +0,0 @@
import sodium, { type StateAddress } from "libsodium-wrappers-sumo";
import { toBase64 } from "./encoding.js";
// The content hash an uploading client records in a file's metadata: unkeyed
// BLAKE2b with a 64-byte output over the original's bytes, fed in chunks, as
// standard base64 with padding. Named after the upstream client's functions.
// The output length is read at call time for the same reason as
// `streamTagFinal` in stream.ts: libsodium sets its constants only once ready.
export const chunkHashInit = (): StateAddress =>
sodium.crypto_generichash_init(null, sodium.crypto_generichash_BYTES_MAX);
export const chunkHashUpdate = (state: StateAddress, chunk: Uint8Array): void =>
sodium.crypto_generichash_update(state, chunk);
export const chunkHashFinal = (state: StateAddress): string =>
toBase64(
sodium.crypto_generichash_final(
state,
sodium.crypto_generichash_BYTES_MAX,
),
);
-1
View File
@@ -7,7 +7,6 @@ export {
} from "./encoding.js";
export { deriveKEK, deriveLoginSubkey } from "./kdf.js";
export { decryptBox, decryptSealed } from "./box.js";
export { chunkHashFinal, chunkHashInit, chunkHashUpdate } from "./hash.js";
export {
decryptBlob,
encryptBlob,
+22 -203
View File
@@ -1,13 +1,8 @@
import { randomBytes } from "node:crypto";
import { readdirSync, rmSync } from "node:fs";
import { randomUUID } from "node:crypto";
import { open, rename, rm } from "node:fs/promises";
import type { FileHandle } from "node:fs/promises";
import { dirname, join } from "node:path";
import { Unzip, UnzipInflate } from "fflate";
import {
chunkHashFinal,
chunkHashInit,
chunkHashUpdate,
fromBase64,
initStreamPull,
pullStreamChunk,
@@ -16,7 +11,6 @@ import {
streamTagFinal,
} from "../crypto/index.js";
import { TruncatedStreamError } from "../errors.js";
import { sanitizeFileName } from "../filename.js";
import { withRetry } from "../retry.js";
import type { ApiClient } from "../api/client.js";
import type { EnteFile } from "../model/types.js";
@@ -103,7 +97,6 @@ const streamDecrypt = async (
onProgress?.(totalPlain);
};
try {
for (;;) {
const { done, value } = await reader.read();
if (value && value.length > 0) {
@@ -113,9 +106,8 @@ const streamDecrypt = async (
while (pendingBytes >= ENC_CHUNK_SIZE) {
const encChunk = takeContiguous(ENC_CHUNK_SIZE);
// A whole chunk that fails to authenticate while the stream
// carries on is corruption, not truncation; that error
// propagates unchanged.
// A whole chunk that fails to authenticate while the stream carries
// on is corruption, not truncation; that error propagates unchanged.
const { plaintext, tag } = pullStreamChunk(state, encChunk);
await consume(plaintext, tag);
}
@@ -125,15 +117,14 @@ const streamDecrypt = async (
const buffer = takeContiguous(pendingBytes);
// Whatever is left over once every whole chunk has been
// consumed must be the stream's final chunk, and a final
// chunk that actually arrived in full authenticates. If
// it does not, the body stopped part-way through a chunk
// — the ordinary shape of a dropped connection. Poly1305
// cannot tell a partial chunk from a corrupt one, so this
// is reported as the truncation it almost always is, with
// the authentication failure kept as the error's cause.
// Only the pull is guarded: a sink failure on a chunk
// that did authenticate is a disk error, not a
// truncation.
// chunk that actually arrived in full authenticates. If it
// does not, the body stopped part-way through a chunk — the
// ordinary shape of a dropped connection. Poly1305 cannot
// tell a partial chunk from a corrupt one, so this is
// reported as the truncation it almost always is, with the
// authentication failure kept as the error's cause. Only the
// pull is guarded: a sink failure on a chunk that did
// authenticate is a disk error, not a truncation.
let pulled;
try {
pulled = pullStreamChunk(state, buffer);
@@ -148,9 +139,6 @@ const streamDecrypt = async (
break;
}
}
} finally {
reader.releaseLock();
}
// Only the last chunk of a secretstream carries TAG_FINAL. Everything a
// dropped connection did deliver still decrypts and authenticates, so the
@@ -169,50 +157,6 @@ const streamDecrypt = async (
return totalPlain;
};
// Fsync a file or a directory, so its contents (for a directory, its entries)
// are on stable storage. Exported for the backup tree's copy, which needs the
// same durability as the writer below.
export const fsyncPath = async (path: string): Promise<void> => {
const handle = await open(path, "r");
try {
await handle.sync();
} finally {
await handle.close();
}
};
// A process-ID check: signal 0 delivers nothing and only reports whether the
// process exists. EPERM means it exists but belongs to another user.
const isRunning = (pid: number): boolean => {
try {
process.kill(pid, 0);
return true;
} catch (err) {
return (err as NodeJS.ErrnoException).code === "EPERM";
}
};
// Delete the temp files a killed process left in `dir`: the writer's
// `.quak-<pid>-<random>.tmp` and the backup copy's
// `.quak-backup-<name>-<pid>-<random>.tmp`. Only files whose process is no
// longer running are removed, so another process writing into the same
// directory keeps its own. A reused process ID can only keep a leftover a while
// longer, never remove a live one.
export const removeLeftoverTempFiles = (dir: string): void => {
let names: string[];
try {
names = readdirSync(dir);
} catch {
return;
}
for (const name of names) {
const match = /^\.quak-(?:.*-)?(\d+)-[0-9a-z]*\.tmp$/.exec(name);
if (match && !isRunning(Number(match[1]))) {
rmSync(join(dir, name), { force: true });
}
}
};
// Stage a write to `destination` atomically and durably, then rename it into
// place. `fill` writes the contents into the open temp file handle — either the
// whole buffer at once (`writeAtomic`) or chunk by chunk as they decrypt
@@ -238,12 +182,8 @@ const stageAtomic = async (
): Promise<void> => {
const dir = dirname(destination);
// The random suffix keeps concurrent downloads of the same destination
// from stepping on each other's temporary file; the process ID lets
// `removeLeftoverTempFiles` tell a leftover from a write in progress.
const tmpPath = join(
dir,
`.quak-${process.pid}-${randomBytes(16).toString("hex")}.tmp`,
);
// from stepping on each other's temporary file.
const tmpPath = join(dir, `.quak-${randomUUID()}.tmp`);
try {
const handle = await open(tmpPath, "w");
try {
@@ -252,15 +192,16 @@ const stageAtomic = async (
} finally {
await handle.close();
}
// `rename` replaces the destination's directory entry rather than
// writing through it: an existing symlink at `destination` is
// replaced, not followed, and the new file has the temp file's
// permissions, not those of the file it replaced.
await rename(tmpPath, destination);
// Fsync the directory so the rename itself survives a crash: renaming
// over a synced temp file still leaves the new directory entry in the
// page cache until the directory is synced.
await fsyncPath(dir);
const dirHandle = await open(dir, "r");
try {
await dirHandle.sync();
} finally {
await dirHandle.close();
}
} catch (err) {
// Best-effort cleanup. A failure to remove the temporary file must
// never replace the error that actually explains what went wrong.
@@ -278,139 +219,31 @@ export const writeAtomic = async (
): Promise<void> =>
stageAtomic(destination, (handle) => handle.writeFile(plaintext));
// Hashes an original's bytes as they are decrypted, for comparison with the
// hash its uploader recorded.
interface ContentHasher {
update: (plaintext: Uint8Array) => void;
digest: () => string;
}
const fileHasher = (): ContentHasher => {
const state = chunkHashInit();
return {
update: (plaintext) => chunkHashUpdate(state, plaintext),
digest: () => chunkHashFinal(state),
};
};
// A live photo is stored as a ZIP of its image and its video, and its recorded
// hash is `<imageHash>:<videoHash>`, each over that part's own bytes. Like the
// upstream client's decoder, this takes the first entries whose names start
// with `image` and `video`.
//
// The ZIP is chosen by its uploader and may expand enormously, so entries are
// hashed as they decompress and never held. fflate's `Unzip` inflates each
// push in one piece, and deflate expands at most about 1000-fold, so the ZIP
// is pushed in 4 KiB slices to keep each decompressed piece near 4 MiB, one
// plaintext chunk. Every entry is started, even one that is not hashed,
// because fflate keeps an unstarted entry's data in memory.
const livePhotoHasher = (fileID: number): ContentHasher => {
const sliceSize = 4096;
const fail = (message: string, cause?: unknown): Error =>
new Error(`download: file ${fileID}: ${message}`, { cause });
const claimed = new Set<string>();
const hashes = new Map<string, string>();
const unzip = new Unzip((entry) => {
const part = ["image", "video"].find((p) => entry.name.startsWith(p));
const target =
part === undefined || claimed.has(part)
? undefined
: { part, state: chunkHashInit() };
if (target !== undefined) claimed.add(target.part);
entry.ondata = (err, data, final) => {
if (err) throw err;
if (target === undefined) return;
chunkHashUpdate(target.state, data);
if (final) hashes.set(target.part, chunkHashFinal(target.state));
};
entry.start();
});
unzip.register(UnzipInflate);
// fflate reports a bad ZIP by throwing, sometimes a TypeError, which the
// retry would take for a network failure; a bad ZIP is never retried.
const push = (data: Uint8Array, final: boolean): void => {
try {
unzip.push(data, final);
} catch (err) {
throw fail("live photo is not a readable ZIP", err);
}
};
return {
update: (plaintext) => {
for (let i = 0; i < plaintext.length; i += sliceSize) {
push(plaintext.subarray(i, i + sliceSize), false);
}
},
digest: () => {
push(new Uint8Array(0), true);
const image = hashes.get("image");
const video = hashes.get("video");
if (image === undefined || video === undefined) {
throw fail(
"live photo ZIP does not hold both an image and a video",
);
}
return `${image}:${video}`;
},
};
};
// Decrypt `stream` straight to `destination`, one plaintext chunk at a time,
// under the atomic writer's temp-then-rename discipline. Memory stays bounded
// by the chunk size: each decrypted chunk is written to the temp file and
// dropped. The rename happens only after the stream authenticates as terminated
// on TAG_FINAL; a truncated stream throws and leaves the destination untouched.
// Returns the plaintext length written.
//
// `original` is the file whose original this is (none for a thumbnail, which
// has no recorded hash). When its metadata has a hash, the decrypted bytes
// must match it or nothing is stored. Both a plain file and a live photo's
// parts are hashed as they stream. The mismatch error is not retried.
const decryptToTemp = async (
destination: string,
stream: ReadableStream<Uint8Array>,
header: Uint8Array,
key: Uint8Array,
onProgress?: ProgressCallback,
original?: EnteFile,
): Promise<number> => {
const expected = original?.metadata.hash;
const hasher =
original === undefined || expected === undefined
? undefined
: original.metadata.fileType === "livePhoto"
? livePhotoHasher(original.id)
: fileHasher();
let bytesWritten = 0;
try {
await stageAtomic(destination, async (handle) => {
bytesWritten = await streamDecrypt(
stream,
header,
key,
async (plaintext) => {
hasher?.update(plaintext);
await handle.write(plaintext);
},
onProgress,
);
if (original === undefined || hasher === undefined) return;
const actual = hasher.digest();
if (actual !== expected) {
throw new Error(
`download: file ${original.id}: content hash ${actual} does not match the hash its uploader recorded, ${expected}`,
);
}
});
} catch (err) {
// Cancel the body so its connection is closed now rather than held
// until the stream is garbage collected. A backup run carries on past
// a failed file, so without this every failure would hold a socket.
// This covers every failure, including a temp file that cannot be
// opened and a header that is rejected before the body is read.
await stream.cancel(err).catch(() => undefined);
throw err;
}
return bytesWritten;
};
@@ -441,18 +274,10 @@ const fetchAndDecrypt = async (
key: Uint8Array,
destination: string,
onProgress?: ProgressCallback,
original?: EnteFile,
): Promise<number> =>
withRetry(async () => {
const stream = await openStream();
return decryptToTemp(
destination,
stream,
header,
key,
onProgress,
original,
);
return decryptToTemp(destination, stream, header, key, onProgress);
}, api.getRetryOptions());
export const downloadFile = async (
@@ -461,10 +286,7 @@ export const downloadFile = async (
outPath?: string,
onProgress?: ProgressCallback,
): Promise<DownloadResult> => {
// `outPath` is the caller's and is used as is; the title is the server's
// and is sanitized so it can only name a file in the current directory.
const resolvedPath =
outPath ?? sanitizeFileName(file.metadata.title, `file-${file.id}`);
const resolvedPath = outPath ?? file.metadata.title;
const header = fromBase64(file.file.decryptionHeader);
const bytesWritten = await fetchAndDecrypt(
api,
@@ -473,7 +295,6 @@ export const downloadFile = async (
file.key,
resolvedPath,
onProgress,
file,
);
return { path: resolvedPath, bytesWritten };
};
@@ -484,9 +305,7 @@ export const downloadThumbnail = async (
outPath?: string,
onProgress?: ProgressCallback,
): Promise<DownloadResult> => {
const resolvedPath =
outPath ??
`thumb_${sanitizeFileName(file.metadata.title, `file-${file.id}`)}`;
const resolvedPath = outPath ?? `thumb_${file.metadata.title}`;
const header = fromBase64(file.thumbnail.decryptionHeader);
const bytesWritten = await fetchAndDecrypt(
api,
-37
View File
@@ -1,37 +0,0 @@
// File names built from server-supplied metadata.
//
// A file's title and a collection's name are decrypted from data the server
// hands us, and quak does not trust the server. Any name taken from them and
// used on disk goes through here, so it can only ever name one file inside the
// directory the caller chose: never a path, never `..`, never hidden, never a
// Windows device name.
//
// A path the user typed (`--out`, `outPath`) is not passed through here: the
// caller is trusted, the server is not.
import { extname } from "node:path";
// Path separators, characters Windows forbids in file names, and control
// characters (NUL included).
// eslint-disable-next-line no-control-regex
const UNSAFE_CHARACTERS = /[/\\:*?"<>|\x00-\x1f\x7f]/g;
// Names Windows reserves for devices, with or without an extension.
const RESERVED_DEVICE_NAME = /^(con|prn|aux|nul|com[1-9]|lpt[1-9])(\.|$)/i;
// `name` made safe to use as a single file name. Each unsafe character becomes
// `_`, a leading run of dots becomes one `_`, and a device name gets a leading
// `_`. A name with none of these comes back unchanged. An empty name becomes
// `fallback`, which the caller derives from the record's ID.
export const sanitizeFileName = (name: string, fallback: string): string => {
if (name === "") return fallback;
const cleaned = name.replace(UNSAFE_CHARACTERS, "_").replace(/^\.+/, "_");
return RESERVED_DEVICE_NAME.test(cleaned) ? `_${cleaned}` : cleaned;
};
// The extension of `title` (".jpg"), or ".bin" when it has none or it holds
// anything but letters and digits.
export const safeExtension = (title: string): string => {
const ext = extname(title);
return /^\.[A-Za-z0-9]+$/.test(ext) ? ext : ".bin";
};
+1 -9
View File
@@ -1,8 +1,4 @@
// package.json is the one place the version is written. tsc copies it to
// dist/package.json, so this path resolves from source and from dist/src/.
import pkg from "../package.json" with { type: "json" };
export const VERSION: string = pkg.version;
export const VERSION = "0.0.0";
export {
Client,
@@ -63,10 +59,6 @@ export {
type EnsureOptions,
type EnsureResult,
type EnsureEvent,
runBackup,
type BackupOptions,
type BackupResult,
type BackupError,
} from "./library/index.js";
export {
RequestPools,
+30 -72
View File
@@ -15,12 +15,12 @@
// Integrity. The reused streaming decrypt is the enforced guarantee: every
// chunk is authenticated and the writer renames the file into place only once
// the stream ends on TAG_FINAL, so a truncated or corrupt fetch throws and
// nothing is stored. For an original whose metadata records a content hash
// (`FileMetadata.hash`), the writer also hashes the decrypted bytes and stores
// nothing if they differ, failing the fetch with an error naming the file. An
// original with no recorded hash is stored unchecked, as the upstream client
// does; thumbnails have none. On top of that this module refuses to record a
// stored file that came out empty.
// nothing is stored. On top of that this module refuses to record a stored file
// that came out empty. The design also asks for a content-hash comparison
// against `FileMetadata.hash` (with a `fileSize` fallback); that is deferred —
// see the PR — because the exact hash construction cannot be confirmed against
// the repo's fixtures and `FileBlob.size` is the encrypted object size, not the
// decrypted length this layer has.
import { existsSync, statSync } from "node:fs";
import {
@@ -39,14 +39,14 @@ import {
downloadFile,
downloadThumbnail,
type ProgressCallback,
removeLeftoverTempFiles,
} from "../download/index.js";
import { safeExtension } from "../filename.js";
import type { EnteFile } from "../model/types.js";
import type { Priority, RequestPools } from "./pools.js";
const DIR_MODE = 0o700;
const FILE_MODE = 0o600;
const TEMP_PREFIX = ".quak-";
const TEMP_SUFFIX = ".tmp";
const GIB = 1024 * 1024 * 1024;
// Owner ruling (#36): bound the originals cache at 100 GiB, but back off when
// the volume has under 50 GiB free so the cache never crowds the disk.
@@ -209,8 +209,10 @@ class AbortDrop extends Error {
}
}
const originalName = (file: EnteFile): string =>
`${file.id}${safeExtension(file.metadata.title)}`;
const originalName = (file: EnteFile): string => {
const ext = extname(file.metadata.title || "") || ".bin";
return `${file.id}${ext}`;
};
// The fileID a cache filename encodes, or undefined when the name is not one
// the cache writes (`<digits><ext>`).
@@ -321,67 +323,17 @@ export class ContentCache implements PhotoContent, ThumbnailsAPI {
return this.get(fileID, "thumbnail", "on-demand", opts?.onProgress);
}
// Get an original for a backup. One not present anywhere is written
// straight to `destination` and recorded there, so no second copy lands
// in the cache; one already present is returned where it is.
async backupOriginal(
fileID: number,
destination: string,
): Promise<ContentResult> {
const result = await this.acquire(
fileID,
"original",
"on-demand",
undefined,
{ destination },
);
return { path: result.path, bytes: result.bytes };
}
async ensure(args: EnsureOptions): Promise<EnsureResult[]> {
return this.ensureThumbnails(args);
}
async ensureThumbnails(args: EnsureOptions): Promise<EnsureResult[]> {
return this.ensureMany(
"thumbnail",
poolPriorityOf(args.priority),
args.fileIDs,
args.signal,
args.onProgress,
);
}
// Fill originals through the content pool for the precache (#48), always at
// background priority so an on-demand `original()` preempts the fill. A
// present original is a map lookup and no fetch; a per-file failure is
// returned, not thrown, so one bad file never halts a background sweep.
async ensureOriginals(args: {
fileIDs: number[];
signal?: AbortSignal;
onProgress?: (event: EnsureEvent) => void;
}): Promise<EnsureResult[]> {
return this.ensureMany(
"original",
"background",
args.fileIDs,
args.signal,
args.onProgress,
);
}
private async ensureMany(
kind: Kind,
priority: Priority,
fileIDs: number[],
signal: AbortSignal | undefined,
onProgress: ((event: EnsureEvent) => void) | undefined,
): Promise<EnsureResult[]> {
const priority = poolPriorityOf(args.priority);
// Dedup the request list so a repeated fileID is fetched once and
// reported once, in first-requested order.
const seen = new Set<number>();
const unique: number[] = [];
for (const id of fileIDs) {
for (const id of args.fileIDs) {
if (!seen.has(id)) {
seen.add(id);
unique.push(id);
@@ -389,20 +341,24 @@ export class ContentCache implements PhotoContent, ThumbnailsAPI {
}
return Promise.all(
unique.map((fileID) =>
this.ensureOne(fileID, kind, priority, signal, onProgress),
this.ensureOne(fileID, priority, args.signal, args.onProgress),
),
);
}
private async ensureOne(
fileID: number,
kind: Kind,
priority: Priority,
signal: AbortSignal | undefined,
onProgress: ((event: EnsureEvent) => void) | undefined,
): Promise<EnsureResult> {
try {
const result = await this.acquire(fileID, kind, priority, signal);
const result = await this.acquire(
fileID,
"thumbnail",
priority,
signal,
);
const status = result.cached ? "skipped" : "done";
onProgress?.({ fileID, status, path: result.path });
return { fileID, path: result.path };
@@ -439,14 +395,13 @@ export class ContentCache implements PhotoContent, ThumbnailsAPI {
// The core: return the cached path if present, else fetch through the pool,
// store, and return it. `cached` distinguishes a present hit (no network,
// no download event) from a fresh fetch. A fetched original is stored at
// `opts.destination` when given, instead of in `originalsDir`.
// no download event) from a fresh fetch.
private async acquire(
fileID: number,
kind: Kind,
priority: Priority,
signal: AbortSignal | undefined,
opts?: { onByte?: ProgressCallback; destination?: string },
opts?: { onByte?: ProgressCallback },
): Promise<{ path: string; bytes: number; cached: boolean }> {
const file = this.getFile(fileID);
if (!file) throw new Error(`content cache: unknown file ${fileID}`);
@@ -487,7 +442,7 @@ export class ContentCache implements PhotoContent, ThumbnailsAPI {
kind === "original" ? this.originalsDir : this.thumbnailsDir;
const dest =
kind === "original"
? (opts?.destination ?? join(dir, originalName(file)))
? join(dir, originalName(file))
: join(dir, `${fileID}${THUMBNAIL_EXT}`);
const pool =
kind === "original" ? this.pools.content : this.pools.thumbnails;
@@ -671,9 +626,6 @@ export class ContentCache implements PhotoContent, ThumbnailsAPI {
}
private async scan(dir: string, into: Map<number, string>): Promise<void> {
// Another process sharing this cache may still be writing its temp
// files, so only those whose process has exited are removed.
removeLeftoverTempFiles(dir);
let entries: string[];
try {
entries = await readdir(dir);
@@ -681,6 +633,12 @@ export class ContentCache implements PhotoContent, ThumbnailsAPI {
return;
}
for (const name of entries) {
if (name.startsWith(TEMP_PREFIX) && name.endsWith(TEMP_SUFFIX)) {
await rm(join(dir, name), { force: true }).catch(
() => undefined,
);
continue;
}
const id = fileIDFromName(name);
const path = join(dir, name);
if (id !== undefined && existsSync(path)) into.set(id, path);
+44 -267
View File
@@ -6,17 +6,10 @@
// existing copy loaded, that first refresh runs in the background and `open()`
// returns as soon as the cached data is ready to serve — a slow or unreachable
// server no longer stalls opening. A background timer then refreshes every
// `refreshIntervalSeconds`. Every default read is answered from RAM — no
// default read touches the network. There is deliberately no `sync()`, no
// `refresh()`, no `serverReachable` flag, and no "before each read" mode
// (design #36).
//
// `fresh()` is the one exception (issue #75, an owner amendment to #36): it
// forces a refresh, awaits it, and only then hands back the read namespaces, so
// a caller that needs server-current data can ask for it. Concurrent `fresh()`
// calls coalesce onto one in-flight refresh, and a refresh that fails rejects
// the caller (the default reads stay silent and serve the last good copy). The
// default methods and the background loop are unchanged.
// `refreshIntervalSeconds`. Every read is answered from RAM — no read touches
// the network. There is deliberately no `sync()`, no `refresh()`, no
// `serverReachable` flag, and no "before each read" mode (design #36): the
// only ways state changes are the refreshes above.
//
// A refresh stages all of its network work first and only mutates the store
// once every fetch has succeeded. A refresh that fails partway therefore never
@@ -26,7 +19,6 @@
// store marked unsaved until a later save actually lands, so a stuck disk is
// never masked by a subsequent empty refresh.
import { rm } from "node:fs/promises";
import { join } from "node:path";
import envPaths from "env-paths";
@@ -48,7 +40,6 @@ import {
type AlbumsAPI,
type PhotosAPI,
type TimelineAPI,
type FreshReads,
} from "./read.js";
import {
ContentCache,
@@ -57,8 +48,6 @@ import {
type EnsureOptions,
type EnsureResult,
} from "./content.js";
import { makeMLDataAPI, type MLDataAPI } from "./mlsearch.js";
import { Precache } from "./precache.js";
export {
Album,
@@ -66,7 +55,6 @@ export {
type AlbumsAPI,
type PhotosAPI,
type TimelineAPI,
type FreshReads,
type PhotoFilter,
type TimelineGroup,
type GroupBy,
@@ -83,43 +71,12 @@ export {
type EnsureResult,
type EnsureEvent,
} from "./content.js";
export { type MLDataAPI, type SimilarResult } from "./mlsearch.js";
import type { CollectionsPage, FilesPage } from "../client.js";
import { MLDATA_BATCH_SIZE, type MLData } from "../mldata-fetch.js";
import type { Collection, EnteFile } from "../model/types.js";
import { runBackup, type BackupOptions, type BackupResult } from "../backup.js";
export {
runBackup,
type BackupOptions,
type BackupResult,
type BackupError,
} from "../backup.js";
export const DEFAULT_REFRESH_INTERVAL_SECONDS = 3;
// The account's cache directory when `cacheDirectory` is not given: the
// env-paths cache directory plus the user id, so each account has its own.
export const defaultCacheDirectory = (userID: number): string =>
join(envPaths("quak", { suffix: "" }).cache, String(userID));
// Project a metadata store into by-id records, filling each record's cache
// paths from the content cache when one is given. Shared by the live read
// projection and the precache's initial seeding at open().
const deriveRecordsFromStore = (
store: MetadataStore,
cache?: ContentCache,
): DerivedRecords => {
const collections = store.listCollections();
const files: EnteFile[] = [];
for (const c of collections) files.push(...store.listFiles(c.id));
return deriveRecords(
collections,
files,
cache ? (fileID) => cache.pathsFor(fileID) : undefined,
);
};
// The slice of `Client` the library depends on. Narrowing to an interface lets
// tests drive a mock with no crypto or network; the real `Client` satisfies it
// structurally.
@@ -148,15 +105,9 @@ export interface LibraryClient {
// A progress event for one unit of background work. A metadata "refresh" or an
// ML "fetchMLData" pass each fire "started" before their network work and then
// exactly one of "done" or "failed"; "failed" carries the error message and
// an ML "done" reports how many payloads it stored. The precache fills
// ("precacheThumbnails"/"precacheOriginals", #48) fire "started"/"done" around
// each sweep that has work, "done" reporting the count newly cached.
// an ML "done" reports how many payloads it stored.
export interface RefreshEvent {
operation:
| "refresh"
| "fetchMLData"
| "precacheThumbnails"
| "precacheOriginals";
operation: "refresh" | "fetchMLData";
status: "started" | "done" | "failed";
error?: string;
fetched?: number;
@@ -186,17 +137,9 @@ export interface LibraryOptions {
// down as the disk fills; `status().originalsLimitBytes` reports it.
cacheOriginalsMaxBytes?: number;
freeBelowBytes?: number;
// An extra pinned predicate OR-ed with the precache's own pinned set
// (favorites + latest week, #48). Pinned originals are never evicted.
// Whether an original is pinned and so never evicted (favorites + latest
// week; the precache unit #48 supplies the set).
isOriginalPinned?: (fileID: number) => boolean;
// The aggressive local precache (#48), all starting inside `open()` with no
// caller input. Thumbnails: every file, newest first, until all are on
// disk. Originals: the favorites album then the latest `precacheOriginalsDays`
// window (the days ending at the newest file). Both default on; the days
// default to 7.
precacheThumbnails?: boolean;
precacheOriginals?: boolean;
precacheOriginalsDays?: number;
}
export interface LibraryStatus {
@@ -222,13 +165,6 @@ export interface LibraryStatus {
// last write or open; undefined when no content cache is open.
originalsUsedBytes?: number;
originalsLimitBytes?: number;
// Precache progress (#48); undefined when no content cache is open. Totals
// are the files targeted (0 when a fill is disabled); "cached" is how many
// of them are on disk.
thumbnailsCached?: number;
thumbnailsTotal?: number;
originalsCached?: number;
originalsPinned?: number;
closed: boolean;
}
@@ -245,10 +181,6 @@ export class Library {
// The thumbnail-prefetch surface (issue #46): drives the thumbnail pool
// with priority, dedup, and abort.
readonly thumbnails: ThumbnailsAPI;
// The content-similarity search surface over the CLIP index (issue #50).
// Present whether or not ML fetching is enabled; with no ML store it
// returns empty results.
readonly mldata: MLDataAPI;
private readonly client: LibraryClient;
private readonly store: MetadataStore;
@@ -260,21 +192,13 @@ export class Library {
private readonly onProgress?: RefreshProgressCallback;
private readonly pools: RequestPools;
// The ML-data cache, present only when the client can fetch ML data.
private readonly mlStore?: MLDataStore;
// The local precache (#48), present only when the content cache is.
private readonly precache?: Precache;
private readonly mldata?: MLDataStore;
private timer?: ReturnType<typeof setTimeout>;
// The in-flight refresh cycle, or undefined when none runs. One slot serves
// both paths: the background loop skips when it is set, and a fresh read
// (issue #75) coalesces onto it or starts one. The promise carries the
// cycle's real outcome (it rejects on failure); the background loop ignores
// that, a fresh read propagates it.
private cycle?: Promise<void>;
private refreshing = false;
// Guards the ML fetch pass so a slow backfill never runs twice at once; a
// refresh whose pass is still running kicks nothing new. Holds the running
// pass, so `close()` can wait for it.
private mlFetch?: Promise<void>;
// refresh whose pass is still running kicks nothing new.
private mlFetching = false;
private closed = false;
private lastRefreshAt?: number;
private lastError?: string;
@@ -302,7 +226,6 @@ export class Library {
pools: RequestPools;
mldata?: MLDataStore;
cache?: ContentCache;
precache?: Precache;
}) {
this.client = args.client;
this.store = args.store;
@@ -312,9 +235,8 @@ export class Library {
this.intervalMs = args.intervalMs;
this.onProgress = args.onProgress;
this.pools = args.pools;
this.mlStore = args.mldata;
this.mldata = args.mldata;
this.cache = args.cache;
this.precache = args.precache;
this.lastRecords = this.deriveNow();
// The read namespaces derive fresh from the store on each call, so they
@@ -335,8 +257,6 @@ export class Library {
return this.cache.ensureThumbnails(opts);
},
};
// Reads the ML store live so results grow as ML data is fetched.
this.mldata = makeMLDataAPI(() => this.mlStore);
}
// Load the cache and start the refresh loop. With an empty cache the first
@@ -349,21 +269,11 @@ export class Library {
static async open(opts: LibraryOptions): Promise<Library> {
const { userID } = opts.client.whoami();
const cacheDirectory =
opts.cacheDirectory ?? defaultCacheDirectory(userID);
const metadataPath = join(cacheDirectory, "metadata.json");
let store = await MetadataStore.load(metadataPath);
// A cache directory given explicitly can hold another account's cache.
// Its records and cursor are not this account's, so delete it and the
// ML data beside it and start empty. A user ID of 0 means the cache
// was never refreshed and so holds nothing to discard.
if (store.userID !== 0 && store.userID !== userID) {
await rm(metadataPath, { force: true });
await rm(join(cacheDirectory, "mldata"), {
recursive: true,
force: true,
});
store = await MetadataStore.load(metadataPath);
}
opts.cacheDirectory ??
join(envPaths("quak", { suffix: "" }).cache, String(userID));
const store = await MetadataStore.load(
join(cacheDirectory, "metadata.json"),
);
const intervalMs =
(opts.refreshIntervalSeconds ?? DEFAULT_REFRESH_INTERVAL_SECONDS) *
1000;
@@ -384,21 +294,7 @@ export class Library {
// the start and the first refresh raises no spurious path-change diff.
const source = opts.contentSource ?? opts.client.contentSource?.();
let cache: ContentCache | undefined;
let precache: Precache | undefined;
if (source) {
// The precache owns the pinned set (favorites + latest week), which
// is the cache's eviction predicate. It is built and seeded from
// the loaded store first so the cache can wire `isPinned` to it, and
// then bound to the cache it fills. A caller-supplied predicate is
// OR-ed in so both survive.
precache = new Precache({
thumbnails: opts.precacheThumbnails,
originals: opts.precacheOriginals,
originalsDays: opts.precacheOriginalsDays,
onEvent: opts.onProgress,
});
precache.update(deriveRecordsFromStore(store));
const extraPinned = opts.isOriginalPinned;
cache = new ContentCache({
pools,
source,
@@ -407,12 +303,9 @@ export class Library {
getFile: (fileID) => store.getFileByID(fileID),
cacheOriginalsMaxBytes: opts.cacheOriginalsMaxBytes,
freeBelowBytes: opts.freeBelowBytes,
isPinned: (fileID) =>
precache!.isPinned(fileID) ||
(extraPinned?.(fileID) ?? false),
isPinned: opts.isOriginalPinned,
});
await cache.open();
precache.bind(cache);
}
const lib = new Library({
@@ -426,15 +319,8 @@ export class Library {
pools,
mldata,
cache,
precache,
});
// Start filling from whatever the loaded store already holds; each
// refresh below re-kicks with the new files (and retries any that
// failed). An empty store starts empty here and fills after its first
// refresh.
precache?.start();
if (store.loadedFromDisk) {
// An existing copy already answers reads; refresh in the background
// and start the interval once that first cycle settles.
@@ -464,13 +350,6 @@ export class Library {
return this.store.getFile(collectionID, fileID);
}
// Any membership of a file, addressed by file id alone. A file's own
// metadata (title, creationTime) is identical across the collections it
// belongs to, so this serves the point commands that hold only a fileID.
getFileByID(fileID: number): EnteFile | undefined {
return this.store.getFileByID(fileID);
}
// A synchronous, RAM-only projection of the whole library into plain
// records (no keys), the surface the GUI reads across IPC. Photos are
// deduplicated to one record per file and ordered newest first.
@@ -499,9 +378,8 @@ export class Library {
for (const c of collections) {
files += this.store.listFiles(c.id).length;
}
const ml = this.mlStore?.stats();
const ml = this.mldata?.stats();
const originals = this.cache?.originalsStatus();
const pre = this.precache?.status();
return {
userID: this.store.userID,
collections: collections.length,
@@ -514,87 +392,18 @@ export class Library {
mlIndexed: ml?.indexed,
originalsUsedBytes: originals?.usedBytes,
originalsLimitBytes: originals?.limitBytes,
thumbnailsCached: pre?.thumbnailsCached,
thumbnailsTotal: pre?.thumbnailsTotal,
originalsCached: pre?.originalsCached,
originalsPinned: pre?.originalsPinned,
closed: this.closed,
};
}
// Fresh reads (issue #75, owner amendment to design #36). Force a refresh,
// wait for it to complete and persist, then hand back the same
// `albums`/`photos`/`timeline` namespaces — now guaranteed to reflect a
// completed server round-trip. Concurrent calls coalesce onto one refresh;
// a refresh that fails rejects here, where the default namespaces would
// instead stay silent and serve the last good copy.
async fresh(): Promise<FreshReads> {
await this.refreshNow();
return {
albums: this.albums,
photos: this.photos,
timeline: this.timeline,
};
}
// Back up every in-scope file to `downloadDirectory` in the historical
// on-disk layout, with a durable failure ledger (issue #51). Waits for a
// completed refresh first, as `fresh()` does, joining one already running,
// and rejects before touching any file when it fails. Then fetches pending
// originals (and optional thumbnails) through the content cache and pools,
// and rebuilds the derived symlink/JSON views from the model. Throws before
// any network work when no download directory is available or no content
// cache backs the originals it must fetch.
backup(opts?: BackupOptions): Promise<BackupResult> {
const downloadDirectory =
opts?.downloadDirectory ?? this.downloadDirectory;
const includeOriginals = opts?.includeOriginals ?? true;
const includeThumbnails = opts?.includeThumbnails ?? false;
if (!downloadDirectory) {
return Promise.reject(
new Error(
"backup requires a downloadDirectory (pass one to " +
"backup() or open the library with one)",
),
);
}
if ((includeOriginals || includeThumbnails) && !this.cache) {
return Promise.reject(
new Error(
"backup requires a library opened with a content cache",
),
);
}
const cache = this.cache;
return runBackup(
{
refresh: () => this.refreshNow(),
listCollections: () => this.store.listCollections(),
listFiles: (id) => this.store.listFiles(id),
original: (fileID, destination) =>
cache!.backupOriginal(fileID, destination),
thumbnail: (fileID) => cache!.thumbnail(fileID),
},
{ ...opts, downloadDirectory },
);
}
// Stop the background timer. Idempotent. An in-flight refresh is left to
// finish; it will not schedule another cycle once closed. The returned
// promise resolves once that refresh (including its cache write), the ML
// fetch pass and the precache fetches already running have all finished,
// so a caller can then remove the cache directory. A refresh failure is
// reported through `status()`, not thrown here.
async close(): Promise<void> {
// finish; it will not schedule another cycle once closed.
close(): void {
this.closed = true;
const precacheClosed = this.precache?.close();
if (this.timer !== undefined) {
clearTimeout(this.timer);
this.timer = undefined;
}
await this.cycle?.catch(() => {});
await this.mlFetch;
await precacheClosed;
}
private scheduleNext(): void {
@@ -606,48 +415,11 @@ export class Library {
this.timer.unref?.();
}
// The background loop's refresh: run a cycle unless one is already in flight
// (or the library is closed), and never let a failure escape — the
// background path reports errors through `status()`/`onProgress`, it does
// not throw. Resolves once the cycle it started (or skipped past) settles.
private runRefresh(): Promise<void> {
if (this.closed || this.cycle) return Promise.resolve();
return this.startCycle().catch(() => {});
}
// A fresh read's refresh (issue #75): force a cycle and await it, rejecting
// if it fails. Concurrent fresh reads coalesce onto the one in-flight cycle
// — the background loop's included — so they never fan out into redundant
// server round-trips.
private refreshNow(): Promise<void> {
if (this.closed) {
return Promise.reject(new Error("the library is closed"));
}
return this.cycle ?? this.startCycle();
}
// Start one refresh cycle and record it as the in-flight cycle so every
// caller coalesces onto it. The returned promise carries the cycle's real
// outcome; each caller attaches the handling its own path needs, and the
// slot is cleared once the cycle settles.
private startCycle(): Promise<void> {
const cycle = this.refreshCycle();
this.cycle = cycle;
void cycle.then(
() => {
if (this.cycle === cycle) this.cycle = undefined;
},
() => {
if (this.cycle === cycle) this.cycle = undefined;
},
);
return cycle;
}
// One refresh cycle: the network fetch and commit, wrapped in the progress
// events and status bookkeeping. Throws when the refresh fails so a fresh
// read can reject; `runRefresh` swallows that throw for the background loop.
private async refreshCycle(): Promise<void> {
// One refresh cycle, guarded so a failure never escapes and overlapping
// cycles never run. Errors are reported, not thrown.
private async runRefresh(): Promise<void> {
if (this.closed || this.refreshing) return;
this.refreshing = true;
this.emit({ operation: "refresh", status: "started" });
try {
await this.refreshOnce();
@@ -658,14 +430,13 @@ export class Library {
// outside the refresh's success/failure so a fetch or disk problem
// there never marks the metadata refresh failed, and it is not
// awaited so it never stalls the refresh interval.
this.mlFetch ??= this.runMLFetch().finally(() => {
this.mlFetch = undefined;
});
void this.runMLFetch();
} catch (err) {
const error = err instanceof Error ? err.message : String(err);
this.lastError = error;
this.emit({ operation: "refresh", status: "failed", error });
throw err;
} finally {
this.refreshing = false;
}
}
@@ -746,12 +517,7 @@ export class Library {
if (change) this.notify(change);
}
this.lastRecords = next;
// Recompute the fill orders and pinned set against the new library.
this.precache?.update(next);
}
// Re-kick the fills every cycle: a finished sweep starts afresh to pick
// up new files and retry any that failed, and a running one is left be.
this.precache?.start();
// Persist whenever RAM holds changes disk has not accepted — including
// changes an earlier cycle staged whose save failed. `unsaved` clears
@@ -770,16 +536,17 @@ export class Library {
// advanced), through the metadata pool, and update the CLIP index. Guarded
// so passes never overlap; a failure is reported, not thrown.
private async runMLFetch(): Promise<void> {
const mldata = this.mlStore;
const mldata = this.mldata;
// Bind so the call keeps the client as its receiver when invoked
// through the pool below.
const fetchMLData = this.client.fetchMLData?.bind(this.client);
if (!mldata || !fetchMLData || this.closed) return;
if (!mldata || !fetchMLData || this.closed || this.mlFetching) return;
const files = this.uniqueFiles();
const needed = mldata.neededFor(files);
if (needed.length === 0) return;
this.mlFetching = true;
this.emit({ operation: "fetchMLData", status: "started" });
try {
const fileKeys = new Map<number, Uint8Array>();
@@ -812,6 +579,8 @@ export class Library {
const error = err instanceof Error ? err.message : String(err);
this.lastMLError = error;
this.emit({ operation: "fetchMLData", status: "failed", error });
} finally {
this.mlFetching = false;
}
}
@@ -844,7 +613,15 @@ export class Library {
// Gather every file membership and project the store into by-id records,
// filling each record's cache paths from the content cache when present.
private deriveNow(): DerivedRecords {
return deriveRecordsFromStore(this.store, this.cache);
const collections = this.store.listCollections();
const files: EnteFile[] = [];
for (const c of collections) files.push(...this.store.listFiles(c.id));
const cache = this.cache;
return deriveRecords(
collections,
files,
cache ? (fileID) => cache.pathsFor(fileID) : undefined,
);
}
private notify(change: LibraryChange): void {
-129
View File
@@ -1,129 +0,0 @@
// The content-similarity search surface over the CLIP index (issue #50).
//
// This is `lib.mldata`. It answers three questions against the ML-data cache
// (#49) without touching the network:
//
// - `forFile` returns the whole stored payload (face boxes, landmarks,
// embedding) for a file, read from disk on demand — the only method here
// that touches the disk, and the only one that is async.
// - `similar` and `searchByEmbedding` rank fileIDs by cosine similarity over
// the packed `Float32Array` index alone. That index (~50k×512) already
// lives in RAM, so each query is a plain loop over it and nothing else.
//
// quak bundles no text encoder (owner-deferred), so `searchByEmbedding` takes
// the query vector the caller has produced elsewhere; `similar` uses the
// query file's own indexed embedding.
import type { MLData } from "../mldata-fetch.js";
import type { MLDataStore, MLIndex } from "./mldata.js";
// How many nearest files a query returns when the caller names no limit.
const DEFAULT_LIMIT = 20;
// One ranked result: a fileID and its cosine similarity to the query, in
// [-1, 1]. Callers wanting only the ids read `.fileID`.
export interface SimilarResult {
fileID: number;
score: number;
}
export interface MLDataAPI {
// The whole stored ML payload for a file, or undefined when it is not
// cached. Reads the payload from disk, so it is async.
forFile(args: { fileID: number }): Promise<MLData | undefined>;
// The files nearest the given file by cosine over their CLIP embeddings,
// most similar first, excluding the file itself. Empty when the file has
// no indexed embedding.
similar(args: { fileID: number; limit?: number }): SimilarResult[];
// The files nearest a caller-supplied query embedding by cosine, most
// similar first. Empty when the query is the wrong length for the index,
// has zero magnitude, or the index is empty.
searchByEmbedding(args: {
embedding: ArrayLike<number>;
limit?: number;
}): SimilarResult[];
}
// Rank the packed index by cosine similarity to `query`, most similar first,
// and return the top `limit`. `skip` (a query file's own id) is left out. Both
// each row's magnitude and the query's are computed here rather than cached:
// the index mutates as ML data is fetched, and one plain pass over ~50k×512
// floats is fast enough that a norm cache would only add a staleness bug. A
// zero-magnitude vector has no direction, so it is dropped rather than divided
// by zero.
const topByCosine = (
index: MLIndex,
query: ArrayLike<number>,
limit: number,
skip?: number,
): SimilarResult[] => {
const { fileIDs, embeddingLength, embeddings } = index;
if (embeddingLength === 0 || query.length !== embeddingLength) return [];
// Every indexed read below is in range: the inner loops run to
// `embeddingLength`, the query is exactly that long (checked above), and
// the packed buffer holds `fileIDs.length * embeddingLength` floats.
// `noUncheckedIndexedAccess` still widens each read to `number | undefined`,
// so they are asserted non-null rather than paying a per-element guard in
// this hot ~50k×512 loop.
let queryNorm = 0;
for (let k = 0; k < embeddingLength; k++) {
const q = query[k]!;
queryNorm += q * q;
}
queryNorm = Math.sqrt(queryNorm);
if (queryNorm === 0) return [];
const results: SimilarResult[] = [];
for (let i = 0; i < fileIDs.length; i++) {
const id = fileIDs[i]!;
if (id === skip) continue;
const base = i * embeddingLength;
let dot = 0;
let norm = 0;
for (let k = 0; k < embeddingLength; k++) {
const v = embeddings[base + k]!;
dot += query[k]! * v;
norm += v * v;
}
if (norm === 0) continue;
results.push({
fileID: id,
score: dot / (queryNorm * Math.sqrt(norm)),
});
}
// Descending score, ties broken by ascending fileID for a stable order.
results.sort((a, b) => b.score - a.score || a.fileID - b.fileID);
return results.slice(0, Math.max(0, Math.trunc(limit)));
};
// Build the search surface over a store the library supplies lazily (the store
// is absent when the client cannot fetch ML data). Reading it per call keeps
// the surface current as the index grows.
export const makeMLDataAPI = (
store: () => MLDataStore | undefined,
): MLDataAPI => ({
forFile: ({ fileID }): Promise<MLData | undefined> => {
const s = store();
return s ? s.readPayload(fileID) : Promise.resolve(undefined);
},
similar: ({ fileID, limit }): SimilarResult[] => {
const s = store();
if (!s) return [];
const index = s.getIndex();
const pos = index.fileIDs.indexOf(fileID);
if (pos < 0) return [];
const base = pos * index.embeddingLength;
const query = index.embeddings.subarray(
base,
base + index.embeddingLength,
);
return topByCosine(index, query, limit ?? DEFAULT_LIMIT, fileID);
},
searchByEmbedding: ({ embedding, limit }): SimilarResult[] => {
const s = store();
if (!s) return [];
return topByCosine(s.getIndex(), embedding, limit ?? DEFAULT_LIMIT);
},
});
-269
View File
@@ -1,269 +0,0 @@
// The aggressive local precache (issue #48), started from `Library.open` with
// no caller input.
//
// Two background fills run concurrently through the shared request pools (#45):
//
// - Thumbnails: every file in the account, newest first, through the
// thumbnail pool until all are on disk. Never evicted. The pool is the same
// one `thumbnails.ensure` uses, so a visible or ahead request always jumps
// ahead of this background fill and a fileID both want is fetched once.
//
// - Originals (the pinned set): through the content pool, the favorites album
// first, then every file whose `takenAt` falls in the latest
// `originalsDays` window — the days ending at the newest file in the
// account. The pinned set is the eviction predicate (#47): a pinned
// original is never evicted, and a file that leaves the set (a favorite
// removed, or the window moving past it on a later refresh) becomes an
// ordinary, evictable original with its bytes left in place.
//
// Both fills yield to on-demand work: every fetch goes to its pool at
// background priority, which the pool serves only after on-demand requests. A
// file already cached costs one map lookup (`pathsFor`) and no fetch. Each
// sweep is driven in bounded chunks so the pool's waiting queue never grows to
// the whole account, keeping on-demand preemption and per-admit cost cheap on a
// large library. Failures are not fatal: an uncached file is retried on the
// next sweep, which `Library` re-kicks after every refresh.
import type { EnsureResult } from "./content.js";
import type { DerivedRecords } from "./records.js";
const ONE_DAY_MS = 24 * 60 * 60 * 1000;
export const DEFAULT_PRECACHE_ORIGINALS_DAYS = 7;
// How many files a sweep submits to a pool before awaiting them. Bounds the
// pool's waiting queue so on-demand work is never stuck behind the whole
// account; the values track each pool's concurrency (#45).
const THUMBNAIL_CHUNK = 25;
const ORIGINAL_CHUNK = 5;
// The precache metrics `Library.status()` surfaces.
export interface PrecacheStatus {
thumbnailsCached: number;
thumbnailsTotal: number;
originalsCached: number;
originalsPinned: number;
}
// A background-fill progress event. Each active sweep fires "started" before
// its fetches and "done" (with the count newly on disk) after; a sweep with
// nothing left to fetch is silent.
export interface PrecacheEvent {
operation: "precacheThumbnails" | "precacheOriginals";
status: "started" | "done";
fetched?: number;
}
// The slice of the content cache the precache drives. The real `ContentCache`
// satisfies it; tests inject a fake.
export interface PrecacheCache {
pathsFor(fileID: number): { originalPath?: string; thumbnailPath?: string };
ensureThumbnails(args: {
fileIDs: number[];
priority: "background";
signal?: AbortSignal;
}): Promise<EnsureResult[]>;
ensureOriginals(args: {
fileIDs: number[];
signal?: AbortSignal;
}): Promise<EnsureResult[]>;
}
export interface PrecacheOptions {
// Default true; false disables the fill and, for originals, the pinning.
thumbnails?: boolean;
originals?: boolean;
// The latest-week window length in days; default 7.
originalsDays?: number;
onEvent?: (event: PrecacheEvent) => void;
}
export class Precache {
private readonly doThumbnails: boolean;
private readonly doOriginals: boolean;
private readonly originalsDays: number;
private readonly onEvent?: (event: PrecacheEvent) => void;
private cache?: PrecacheCache;
// Every file in the account, newest first (the thumbnail fill order).
private thumbOrder: number[] = [];
// The pinned originals in fetch order: favorites first, then the window.
private originalsOrder: number[] = [];
private pinned = new Set<number>();
// A sweep runs at most once per fill at a time; a re-kick while one runs is
// a no-op, and the next refresh re-kicks after it finishes. Each holds the
// running sweep, so `close()` can wait for it.
private thumbSweep?: Promise<void>;
private originalsSweep?: Promise<void>;
private readonly aborter = new AbortController();
private closed = false;
constructor(opts: PrecacheOptions = {}) {
this.doThumbnails = opts.thumbnails ?? true;
this.doOriginals = opts.originals ?? true;
this.originalsDays =
opts.originalsDays ?? DEFAULT_PRECACHE_ORIGINALS_DAYS;
this.onEvent = opts.onEvent;
}
// Attach the cache the fills fetch through. `isPinned` works before this is
// called, so the cache can be constructed with `isPinned` wired in and then
// bound here.
bind(cache: PrecacheCache): void {
this.cache = cache;
}
// Whether an original is pinned (favorites + the latest-week window), the
// eviction predicate (#47). False for every file when originals precaching
// is disabled.
isPinned(fileID: number): boolean {
return this.pinned.has(fileID);
}
// Recompute the fill orders and the pinned set from the current projection.
// Called at open and after every refresh that changes the library.
update(records: DerivedRecords): void {
const photos = [...records.photos.values()].sort(
(a, b) => b.takenAt - a.takenAt || b.fileID - a.fileID,
);
this.thumbOrder = this.doThumbnails ? photos.map((p) => p.fileID) : [];
const pinned = new Set<number>();
const order: number[] = [];
if (this.doOriginals) {
// Favorites first, in the album's own newest-first order.
for (const album of records.albums.values()) {
if (album.type !== "favorites") continue;
for (const id of album.fileIDs) {
if (!pinned.has(id)) {
pinned.add(id);
order.push(id);
}
}
}
// Then the latest-week window, ending at the newest file. Photos
// are newest first, so stop once one falls before the window start.
const newest = photos[0];
if (newest !== undefined) {
const windowStart =
newest.takenAt - this.originalsDays * ONE_DAY_MS;
for (const p of photos) {
if (p.takenAt < windowStart) break;
if (!pinned.has(p.fileID)) {
pinned.add(p.fileID);
order.push(p.fileID);
}
}
}
}
this.pinned = pinned;
this.originalsOrder = order;
}
status(): PrecacheStatus {
const cache = this.cache;
let thumbnailsCached = 0;
let originalsCached = 0;
if (cache) {
for (const id of this.thumbOrder)
if (cache.pathsFor(id).thumbnailPath !== undefined)
thumbnailsCached++;
for (const id of this.originalsOrder)
if (cache.pathsFor(id).originalPath !== undefined)
originalsCached++;
}
return {
thumbnailsCached,
thumbnailsTotal: this.thumbOrder.length,
originalsCached,
originalsPinned: this.originalsOrder.length,
};
}
// Kick both fills. Idempotent: a fill already sweeping is left alone. Safe
// to call after every refresh; a finished fill starts a fresh sweep that
// picks up new files and retries any that failed before.
start(): void {
if (this.closed || !this.cache) return;
if (this.doThumbnails) this.kickThumbnails();
if (this.doOriginals) this.kickOriginals();
}
// Stop the fills. In-flight fetches are left to settle; queued ones drop.
// Resolves once both sweeps have finished, so nothing is still writing.
async close(): Promise<void> {
this.closed = true;
this.aborter.abort();
await Promise.all([
this.thumbSweep?.catch(() => {}),
this.originalsSweep?.catch(() => {}),
]);
}
private kickThumbnails(): void {
if (this.thumbSweep) return;
this.thumbSweep = this.sweep(
"precacheThumbnails",
() => this.thumbOrder,
(id) => this.cache!.pathsFor(id).thumbnailPath !== undefined,
THUMBNAIL_CHUNK,
(chunk) =>
this.cache!.ensureThumbnails({
fileIDs: chunk,
priority: "background",
signal: this.aborter.signal,
}),
).finally(() => {
this.thumbSweep = undefined;
});
}
private kickOriginals(): void {
if (this.originalsSweep) return;
this.originalsSweep = this.sweep(
"precacheOriginals",
() => this.originalsOrder,
(id) => this.cache!.pathsFor(id).originalPath !== undefined,
ORIGINAL_CHUNK,
(chunk) =>
this.cache!.ensureOriginals({
fileIDs: chunk,
signal: this.aborter.signal,
}),
).finally(() => {
this.originalsSweep = undefined;
});
}
// One fill sweep: skip files already on disk (one lookup each), fetch the
// rest in bounded chunks, and report progress only when there was work.
private async sweep(
operation: PrecacheEvent["operation"],
order: () => number[],
present: (fileID: number) => boolean,
chunkSize: number,
fetch: (chunk: number[]) => Promise<EnsureResult[]>,
): Promise<void> {
const todo = order().filter((id) => !present(id));
if (todo.length === 0) return;
this.emit({ operation, status: "started" });
let fetched = 0;
for (let i = 0; i < todo.length && !this.closed; i += chunkSize) {
const results = await fetch(todo.slice(i, i + chunkSize));
for (const r of results) if (r.path !== undefined) fetched++;
}
this.emit({ operation, status: "done", fetched });
}
private emit(event: PrecacheEvent): void {
if (!this.onEvent) return;
// A misbehaving callback must not break the fill loop.
try {
this.onEvent(event);
} catch {
// ignore
}
}
}
-9
View File
@@ -193,15 +193,6 @@ export interface TimelineAPI {
groups(args: { groupBy: GroupBy; filter?: PhotoFilter }): TimelineGroup[];
}
// The surface `Library.fresh()` resolves to (issue #75). It is the same three
// read namespaces as the default `albums`/`photos`/`timeline`, handed back only
// after a forced refresh has brought the local copy current.
export interface FreshReads {
albums: AlbumsAPI;
photos: PhotosAPI;
timeline: TimelineAPI;
}
export const makeAlbumsAPI = (
derive: () => DerivedRecords,
content?: PhotoContent,
+61 -119
View File
@@ -1,15 +1,16 @@
import { mkdirSync, readFileSync, writeFileSync } from "node:fs";
import {
mkdirSync,
mkdtempSync,
readFileSync,
rmSync,
writeFileSync,
} from "node:fs";
import { join } from "node:path";
import { tmpdir } from "node:os";
import * as jpeg from "jpeg-js";
import exifReader from "exif-reader";
import type { Client } from "./client.js";
import type { Library, Photo } from "./library/index.js";
import { sanitizeFileName } from "./filename.js";
import {
fetchMLDataBatch,
MLDATA_BATCH_SIZE,
type MLData,
} from "./mldata-fetch.js";
import { fetchMLData } from "./mldata-fetch.js";
import type { EnteFile } from "./model/types.js";
export type ProgressCallback = (message: string) => void;
@@ -19,66 +20,45 @@ export interface MetadataBackupOptions {
onProgress?: ProgressCallback;
}
// Find the raw EXIF APP1 segment in JPEG bytes. Returns `exif` (the segment
// data, starting at the "Exif\0\0" header) when there is one, nothing when the
// bytes are not a JPEG or carry no EXIF, and `error` when the segment layout is
// malformed. Each segment length is checked against the bytes that remain and
// each step moves forward by at least 4 bytes, so the scan ends on any input.
export const extractExifFromJpeg = (
buf: Uint8Array,
): { exif?: Buffer; error?: string } => {
if (buf[0] !== 0xff || buf[1] !== 0xd8) return {};
const sanitizePath = (name: string): string =>
name.replace(/[/\\:*?"<>|]/g, "_").replace(/^\.+/, "_");
// Extract the raw EXIF APP1 segment from JPEG bytes. Returns the EXIF
// data buffer (starting after the APP1 length field, at the "Exif\0\0"
// header) or undefined if no APP1 marker is found.
const extractExifFromJpeg = (buf: Uint8Array): Buffer | undefined => {
if (buf[0] !== 0xff || buf[1] !== 0xd8) return undefined;
let offset = 2;
while (offset < buf.length) {
if (offset + 2 > buf.length)
return { error: `truncated segment marker at byte ${offset}` };
if (buf[offset] !== 0xff)
return { error: `no segment marker at byte ${offset}` };
while (offset < buf.length - 1) {
if (buf[offset] !== 0xff) return undefined;
const marker = buf[offset + 1]!;
if (marker === 0xda) return {}; // start of scan, no more markers
if (offset + 4 > buf.length)
return { error: `truncated segment length at byte ${offset}` };
if (marker === 0xda) break; // start of scan, no more markers
if (offset + 3 >= buf.length) break;
const len = (buf[offset + 2]! << 8) | buf[offset + 3]!;
// The length counts its own two bytes, so anything under 2 is invalid.
if (len < 2)
return {
error: `segment length ${len} at byte ${offset} is too small`,
};
if (offset + 2 + len > buf.length)
return {
error: `segment length ${len} at byte ${offset} runs past the end of the file`,
};
if (marker === 0xe1) {
// APP1 — check for "Exif\0\0" header. A length under 8 cannot hold
// the six-byte header, so the segment is not EXIF; below 6 the
// bytes compared would also lie past the segment.
// APP1 — check for "Exif\0\0" header
if (
len >= 8 &&
buf[offset + 4] === 0x45 &&
buf[offset + 5] === 0x78 &&
buf[offset + 6] === 0x69 &&
buf[offset + 7] === 0x66
) {
return {
exif: Buffer.from(
return Buffer.from(
buf.buffer,
buf.byteOffset + offset + 4,
len - 2,
),
};
);
}
}
offset += 2 + len;
}
return { error: "file ends before the image data" };
return undefined;
};
// Extract dimensions, EXIF and XMP from a file's bytes. When the EXIF segment
// is malformed or cannot be parsed, the record carries the reason in
// `exifError`.
export const extractImageMetadata = (
const extractImageMetadata = (
fileBytes: Uint8Array,
): Record<string, unknown> | undefined => {
try {
const result: Record<string, unknown> = {};
// Try to get dimensions from JPEG decode
@@ -91,19 +71,15 @@ export const extractImageMetadata = (
result.width = decoded.width;
result.height = decoded.height;
} catch {
// Not every original is a JPEG (PNG, HEIC, video), so a failed decode
// is expected and only means no dimensions; a malformed JPEG is still
// reported below through `exifError`.
// Not a JPEG or corrupt; still try EXIF extraction
}
const { exif, error } = extractExifFromJpeg(fileBytes);
if (error) result.exifError = error;
if (exif) {
const exifBuf = extractExifFromJpeg(fileBytes);
if (exifBuf) {
try {
result.exif = exifReader(exif);
} catch (err) {
result.exifRaw = exif.toString("base64");
result.exifError = err instanceof Error ? err.message : String(err);
result.exif = exifReader(exifBuf);
} catch {
result.exifRaw = exifBuf.toString("base64");
}
}
@@ -123,32 +99,33 @@ export const extractImageMetadata = (
}
return Object.keys(result).length > 0 ? result : undefined;
} catch {
return undefined;
}
};
// Read a file's original bytes through the library's content cache and extract
// its embedded image metadata. The bytes come from `photo.original()` — the
// same on-disk cache the rest of the library fills — rather than a fresh
// per-call download to a throwaway temp file.
const extractExif = async (
photo: Photo,
client: Client,
file: EnteFile,
): Promise<Record<string, unknown> | undefined> => {
const { path } = await photo.original();
const fileBytes = new Uint8Array(readFileSync(path));
const tmpDir = mkdtempSync(join(tmpdir(), "quak-exif-"));
try {
const origPath = join(tmpDir, "original");
await client.downloadFile(file, origPath);
const fileBytes = new Uint8Array(readFileSync(origPath));
return extractImageMetadata(fileBytes);
} catch {
return undefined;
} finally {
rmSync(tmpDir, { recursive: true, force: true });
}
};
// Dump every decrypted metadata layer the account holds into a directory tree
// of plain JSON: account, per-collection, and per-file records including the
// private and public magic metadata and (by default) the ML data. Collections
// and files are enumerated from the library's cache, which the caller refreshes
// first. Returns how many ML data requests failed; their files are still
// written, with `mlDataError` in place of `mlData`.
export const runMetadataBackup = async (
lib: Library,
client: Client,
outDir: string,
opts?: MetadataBackupOptions,
): Promise<{ failedMLBatches: number }> => {
): Promise<void> => {
const log = opts?.onProgress ?? (() => {});
const wantExif = opts?.exif ?? false;
@@ -162,20 +139,14 @@ export const runMetadataBackup = async (
);
log("Fetching collections...");
const collections = await client.listCollections();
// Enumerate through the library's read surface. Each album carries its
// photos, but the full decrypted `Collection`/`EnteFile` records (with the
// magic-metadata layers this dump exists to preserve) come from the
// library's by-id accessors.
const allFiles: { file: EnteFile; photo: Photo; colDirName: string }[] = [];
const allFiles: { file: EnteFile; colDirName: string }[] = [];
const fileKeys = new Map<number, Uint8Array>();
const seenFileIDs = new Set<number>();
for (const album of lib.albums.list()) {
const col = lib.getCollection(album.collectionID);
if (!col) continue;
const dirName = `${col.id}-${sanitizeFileName(col.name, "unnamed")}`;
for (const col of collections) {
const dirName = `${col.id}-${sanitizePath(col.name || "unnamed")}`;
const colDir = join(outDir, "collections", dirName);
mkdirSync(colDir, { recursive: true });
@@ -199,13 +170,11 @@ export const runMetadataBackup = async (
);
log(`[${col.name}] Fetching files...`);
const photos = album.photos.list();
log(`[${col.name}] ${photos.length} file(s)`);
const files = await client.listFiles(col.id, col.key);
log(`[${col.name}] ${files.length} file(s)`);
for (const photo of photos) {
const file = lib.getFile(col.id, photo.fileID);
if (!file) continue;
allFiles.push({ file, photo, colDirName: dirName });
for (const file of files) {
allFiles.push({ file, colDirName: dirName });
if (!seenFileIDs.has(file.id)) {
fileKeys.set(file.id, file.key);
seenFileIDs.add(file.id);
@@ -213,35 +182,16 @@ export const runMetadataBackup = async (
}
}
// One failed request (retries exhausted) must not end the dump: its files
// get the reason in `mlDataError` and the other batches go on.
log("Fetching ML data (face detections, CLIP embeddings)...");
const mlDataMap = new Map<number, MLData>();
const mlDataErrors = new Map<number, string>();
let failedMLBatches = 0;
const fileIDs = [...fileKeys.keys()];
for (let i = 0; i < fileIDs.length; i += MLDATA_BATCH_SIZE) {
const batch = fileIDs.slice(i, i + MLDATA_BATCH_SIZE);
try {
const result = await fetchMLDataBatch(
const mlDataMap = await fetchMLData(
client.getApiClient(),
batch,
[...fileKeys.keys()],
fileKeys,
);
for (const [id, payload] of result) mlDataMap.set(id, payload);
} catch (err) {
const reason = err instanceof Error ? err.message : String(err);
failedMLBatches++;
log(
`ML data request for ${batch.length} file(s) failed: ${reason}`,
);
for (const id of batch) mlDataErrors.set(id, reason);
}
}
log(`Got ML data for ${mlDataMap.size} file(s)`);
const writtenFileIDs = new Set<number>();
for (const { file, photo, colDirName } of allFiles) {
for (const { file, colDirName } of allFiles) {
const colDir = join(outDir, "collections", colDirName);
const fileMeta: Record<string, unknown> = {
@@ -257,18 +207,11 @@ export const runMetadataBackup = async (
const ml = mlDataMap.get(file.id);
if (ml) fileMeta.mlData = ml;
const mlError = mlDataErrors.get(file.id);
if (mlError) fileMeta.mlDataError = mlError;
if (wantExif && !writtenFileIDs.has(file.id)) {
log(`[${file.metadata.title}] Extracting EXIF...`);
try {
const exifData = await extractExif(photo);
const exifData = await extractExif(client, file);
if (exifData) fileMeta.imageMetadata = exifData;
} catch (err) {
fileMeta.imageMetadataError =
err instanceof Error ? err.message : String(err);
}
}
writtenFileIDs.add(file.id);
@@ -279,5 +222,4 @@ export const runMetadataBackup = async (
}
log("Metadata backup complete.");
return { failedMLBatches };
};
+25 -2
View File
@@ -5,8 +5,9 @@
// comes back encrypted under the file's own key and gzipped; decrypting and
// gunzipping yields the JSON payload
// `{ face: { faces: [...] }, clip: { embedding } }`. Ente caps a request at 200
// ids, so callers that want many at once split them into batches of
// `MLDATA_BATCH_SIZE` and call `fetchMLDataBatch` once per batch.
// ids, so `fetchMLData` batches for callers that want many at once while
// `fetchMLDataBatch` is the single-request unit the library submits to its
// request pool.
import { gunzipSync } from "node:zlib";
@@ -68,3 +69,25 @@ export const fetchMLDataBatch = async (
}
return result;
};
// Fetch ML data for arbitrarily many ids, batching at `MLDATA_BATCH_SIZE`. Used
// by the one-shot metadata backup; the library fetches through its request pool
// with `fetchMLDataBatch` instead.
export const fetchMLData = async (
api: ApiClient,
fileIDs: number[],
fileKeys: Map<number, Uint8Array>,
): Promise<Map<number, MLData>> => {
const result = new Map<number, MLData>();
for (let i = 0; i < fileIDs.length; i += MLDATA_BATCH_SIZE) {
const batch = fileIDs.slice(i, i + MLDATA_BATCH_SIZE);
for (const [id, payload] of await fetchMLDataBatch(
api,
batch,
fileKeys,
)) {
result.set(id, payload);
}
}
return result;
};
+2 -33
View File
@@ -34,28 +34,6 @@ const FILE_TYPE_MAP: Record<number, FileType> = {
const parseFileType = (n: number): FileType => FILE_TYPE_MAP[n] ?? "unknown";
// The hash the uploading client recorded for the original's bytes, read the
// way the upstream client's `metadataHash` reads it: `hash` if present,
// otherwise, for a live photo from an older client that wrote the two parts
// separately, `<imageHash>:<videoHash>`. A field that is not a non-empty
// string counts as absent, and a file with no hash at all is normal.
const expectedHash = (json: Record<string, unknown>): string | undefined => {
const text = (v: unknown): string | undefined =>
typeof v === "string" && v !== "" ? v : undefined;
const hash = text(json.hash);
if (hash !== undefined) return hash;
const imageHash = text(json.imageHash);
const videoHash = text(json.videoHash);
if (
json.fileType === 2 &&
imageHash !== undefined &&
videoHash !== undefined
) {
return `${imageHash}:${videoHash}`;
}
return undefined;
};
export const decryptCollection = (
raw: RawCollection,
keys: KeyMaterial,
@@ -120,24 +98,15 @@ export const decryptFile = (
key,
);
const metadataJSON = JSON.parse(new TextDecoder().decode(metadataBytes));
if (
typeof metadataJSON !== "object" ||
metadataJSON === null ||
Array.isArray(metadataJSON)
) {
throw new Error(`file ${raw.id}: metadata is not a JSON object`);
}
const metadata: FileMetadata = {
// The server controls this JSON: a title that is missing or not a
// string becomes "", never an arbitrary value.
title: typeof metadataJSON.title === "string" ? metadataJSON.title : "",
title: metadataJSON.title ?? "",
fileType: parseFileType(metadataJSON.fileType ?? -1),
creationTime: metadataJSON.creationTime ?? 0,
modificationTime: metadataJSON.modificationTime ?? 0,
latitude: metadataJSON.latitude,
longitude: metadataJSON.longitude,
hash: expectedHash(metadataJSON),
hash: metadataJSON.hash,
};
const magicMetadata = decryptMagicMetadata(raw.magicMetadata, key);
-3
View File
@@ -29,9 +29,6 @@ export interface FileMetadata {
modificationTime: Microseconds;
latitude?: number;
longitude?: number;
// The content hash the uploader recorded (see `expectedHash` in
// decrypt.ts); `downloadFile` refuses an original that does not match it.
// Absent for files from very old clients.
hash?: string;
}
+12 -25
View File
@@ -88,21 +88,18 @@ const MAX_CAUSE_DEPTH = 8;
// errno on the error it throws — it hangs the underlying socket error off
// `cause`, sometimes more than one level down — so a classifier that only read
// the top-level error would see a bare `Error` and call every dropped
// connection permanent. `complete` is false when the walk stopped at the
// depth limit with more of the chain still below it.
const causeCodes = (err: unknown): { codes: string[]; complete: boolean } => {
// connection permanent.
const causeCodes = (err: unknown): string[] => {
const codes: string[] = [];
let current: unknown = err;
for (let depth = 0; depth < MAX_CAUSE_DEPTH; depth++) {
if (current === null || typeof current !== "object") {
return { codes, complete: true };
}
if (current === null || typeof current !== "object") break;
const { code, cause } = current as { code?: unknown; cause?: unknown };
if (typeof code === "string") codes.push(code);
if (cause === current) return { codes, complete: true };
if (cause === current) break;
current = cause;
}
return { codes, complete: current === null || typeof current !== "object" };
return codes;
};
const isAbort = (err: unknown): boolean => {
@@ -148,15 +145,15 @@ export const isRetryable = (err: unknown): boolean => {
// have succeeded; the cost of the imprecision is bounded by the attempt
// count.
if (err instanceof TypeError) return true;
return causeCodes(err).codes.some((code) => TRANSPORT_CODES.has(code));
return causeCodes(err).some((code) => TRANSPORT_CODES.has(code));
};
// Could the first attempt already have taken effect on the server?
//
// `isRetryable` is the wrong question for a request that changes state.
// `postJSON` and `putJSON` use this for every `POST` and `PUT` listed in the
// README under "Endpoints used"; verifying a second factor, for one, consumes
// one of a small number of attempts. They are replayed only on the failures in
// quak's non-idempotent calls are `/users/srp/create-session`,
// `/users/two-factor/verify` — which consumes one of a small number of 2FA
// attempts — and `/files/thumbnail`. They are replayed only on the failures in
// `CONNECT_CODES`, which establish that no TCP connection to the server ever
// existed: there was no address to connect to, or the peer refused the
// connection outright. A request byte cannot have been transmitted, so the
@@ -165,19 +162,9 @@ export const isRetryable = (err: unknown): boolean => {
// Everything else is ambiguous. A 5xx proves the server did process the
// request. A reset or a broken pipe can arrive after it was fully sent and
// acted on. A routing errno can be delivered on an established socket. A
// deadline says nothing at all about the server's state. So every errno in the
// cause chain must be a connect errno: one other errno anywhere in the chain
// is doubt, and doubt is not replayed. A chain longer than the walk is doubt
// too: the links below the limit were never read.
export const isSafeToReplay = (err: unknown): boolean => {
const { codes, complete } = causeCodes(err);
return (
isRetryable(err) &&
complete &&
codes.length > 0 &&
codes.every((code) => CONNECT_CODES.has(code))
);
};
// deadline says nothing at all about the server's state.
export const isSafeToReplay = (err: unknown): boolean =>
isRetryable(err) && causeCodes(err).some((code) => CONNECT_CODES.has(code));
export interface WithRetryOptions extends RetryOptions {
isRetryable?: (err: unknown) => boolean;
+86 -201
View File
@@ -2,26 +2,16 @@ import { createHash } from "node:crypto";
import { readFileSync } from "node:fs";
import * as jpeg from "jpeg-js";
import type { Client } from "./client.js";
import type { Library } from "./library/index.js";
import { ApiError } from "./api/client.js";
import { encryptBlob, toBase64 } from "./crypto/index.js";
import { downloadFile } from "./download/index.js";
import type { EnteFile } from "./model/types.js";
import { mkdtempSync, rmSync } from "node:fs";
import { join } from "node:path";
import { tmpdir } from "node:os";
// The server refuses a thumbnail larger than the one it already records for the
// file (`thumbnail.size`, the encrypted size), so these encodings are tried
// from largest to smallest and the first that fits is uploaded.
const THUMB_ENCODINGS = [
{ maxDimension: 720, quality: 50 },
{ maxDimension: 720, quality: 30 },
{ maxDimension: 480, quality: 30 },
{ maxDimension: 320, quality: 20 },
{ maxDimension: 160, quality: 20 },
];
// The server accepts a new thumbnail only from the file's owner, so files other
// people own in albums shared with this account are never checked or repaired.
const NOT_OWNED_REASON =
"owned by another account (only the owner can replace its thumbnail)";
const THUMB_MAX_DIMENSION = 720;
const THUMB_JPEG_QUALITY = 50;
export interface MissingThumbnailInfo {
fileID: number;
@@ -30,61 +20,34 @@ export interface MissingThumbnailInfo {
reason: string;
}
// Three outcomes, not two. "fixed": a thumbnail was generated and uploaded.
// "failed": something went wrong (download, encode, upload) and the file still
// has no thumbnail. "skipped": the server would refuse any thumbnail for the
// file or this helper cannot regenerate it — a file another account owns, a
// recorded thumbnail size nothing fits within, a video, or an image that is
// not a baseline JPEG. Skipped is a deliberate, expected outcome, not an error
// (issue #17): the repair path is JPEG-only because `jpeg-js` is, and a PNG or
// HEIC is left for a format-aware tool rather than reported as a failure.
export type ThumbnailFixStatus = "fixed" | "skipped" | "failed";
export interface ThumbnailFixResult {
fileID: number;
title: string;
collection: string;
status: ThumbnailFixStatus;
// Why the file was skipped or failed; unset when it was fixed.
reason?: string;
success: boolean;
error?: string;
}
export type ProgressCallback = (message: string) => void;
// Enumerate every file the library knows about, newest album first, each file
// once, and report those whose server-side thumbnail is missing. "Missing" is
// only two answers: an empty body, or a 404. Any other error reaching this
// point has already exhausted its retries — a failing server, a dropped
// connection, a deadline — and says nothing about whether the thumbnail
// exists, so it is logged and the file is left unreported. That distinction is
// what stops `fix-missing-thumbnails` from regenerating and uploading over
// thumbnails that were fine all along while the CDN was briefly returning 500s.
// Files another account owns are logged as skipped and not checked.
export const listMissingThumbnails = async (
lib: Library,
client: Client,
onProgress?: ProgressCallback,
): Promise<MissingThumbnailInfo[]> => {
const log = onProgress ?? (() => {});
const api = client.getApiClient();
const { userID } = client.whoami();
const missing: MissingThumbnailInfo[] = [];
const seen = new Set<number>();
for (const album of lib.albums.list()) {
log(`[${album.name}] Checking thumbnails...`);
for (const photo of album.photos.list()) {
if (seen.has(photo.fileID)) continue;
seen.add(photo.fileID);
const file = lib.getFile(album.collectionID, photo.fileID);
if (file && file.ownerID !== userID) {
log(
`[${album.name}] Skipping ${photo.title}: ${NOT_OWNED_REASON}`,
);
continue;
}
const collections = await client.listCollections();
for (const col of collections) {
log(`[${col.name}] Checking thumbnails...`);
const files = await client.listFiles(col.id, col.key);
for (const file of files) {
if (seen.has(file.id)) continue;
seen.add(file.id);
try {
const stream = await api.getThumbnailStream(photo.fileID);
const api = client.getApiClient();
const stream = await api.getThumbnailStream(file.id);
const reader = stream.getReader();
let totalBytes = 0;
for (;;) {
@@ -94,23 +57,35 @@ export const listMissingThumbnails = async (
}
if (totalBytes === 0) {
missing.push({
fileID: photo.fileID,
title: photo.title,
collection: album.name,
fileID: file.id,
title: file.metadata.title,
collection: col.name,
reason: "empty thumbnail (0 bytes)",
});
}
} catch (err) {
// A 404 is the server stating the thumbnail is not there:
// that, and an empty body, are the only two answers that mean
// "missing". Anything else reaching this point is a failure
// that already exhausted its retries — a failing server, a
// dropped connection, a deadline — and says nothing about
// whether the thumbnail exists.
//
// The distinction is what stops `helper
// fix-missing-thumbnails` from downloading originals,
// regenerating thumbnails and uploading them over thumbnails
// that were fine all along, because the CDN was briefly
// returning 500s while this ran.
if (err instanceof ApiError && err.status === 404) {
missing.push({
fileID: photo.fileID,
title: photo.title,
collection: album.name,
fileID: file.id,
title: file.metadata.title,
collection: col.name,
reason: "thumbnail not found (HTTP 404)",
});
} else {
log(
`[${album.name}] Could not check ${photo.title}: ${err instanceof Error ? err.message : String(err)} (not reported as missing)`,
`[${col.name}] Could not check ${file.metadata.title}: ${err instanceof Error ? err.message : String(err)} (not reported as missing)`,
);
}
}
@@ -119,7 +94,7 @@ export const listMissingThumbnails = async (
return missing;
};
// Bilinear resize of an RGBA pixel buffer.
// Bilinear resize of RGBA pixel buffer
const resizeRGBA = (
src: Uint8Array,
srcW: number,
@@ -158,13 +133,17 @@ const resizeRGBA = (
return dst;
};
const generateThumbnail = (
decoded: { data: Uint8Array; width: number; height: number },
maxDimension: number,
quality: number,
): Uint8Array => {
const generateThumbnail = (fileBytes: Uint8Array): Uint8Array => {
const decoded = jpeg.decode(fileBytes, {
useTArray: true,
formatAsRGBA: true,
});
const { width: srcW, height: srcH } = decoded;
const scale = Math.min(maxDimension / srcW, maxDimension / srcH, 1);
const scale = Math.min(
THUMB_MAX_DIMENSION / srcW,
THUMB_MAX_DIMENSION / srcH,
1,
);
const dstW = Math.round(srcW * scale);
const dstH = Math.round(srcH * scale);
@@ -177,59 +156,12 @@ const generateThumbnail = (
const encoded = jpeg.encode(
{ data: pixels, width: dstW, height: dstH },
quality,
THUMB_JPEG_QUALITY,
);
return new Uint8Array(encoded.data);
};
// A baseline/JFIF JPEG starts with the SOI marker 0xFFD8. `jpeg-js` decodes
// only JPEG, so this signature check is what separates a file the helper can
// regenerate from one it must skip: a PNG, HEIC, or the odd non-image byte
// stream all fail this and are reported as skipped rather than crashing the
// decoder into an opaque failure (issue #17).
const isJpeg = (bytes: Uint8Array): boolean =>
bytes.length >= 2 && bytes[0] === 0xff && bytes[1] === 0xd8;
// The reason a file cannot have a JPEG thumbnail regenerated for it, known from
// its record alone before any bytes are fetched, or undefined when it might. A
// still image still has to be checked against its actual bytes once
// downloaded.
const reasonToSkip = (file: EnteFile, userID: number): string | undefined => {
if (file.ownerID !== userID) {
return NOT_OWNED_REASON;
}
if (file.metadata.fileType !== "image") {
return `unsupported file type: ${file.metadata.fileType} (only JPEG images can be regenerated)`;
}
if (!file.thumbnail.size) {
return `recorded thumbnail size is ${file.thumbnail.size ?? "unknown"} (the server refuses a thumbnail larger than the one it records)`;
}
return undefined;
};
// Encrypt the largest encoding of the decoded image whose ciphertext is no
// larger than `maxSize`, or return undefined when even the smallest is larger.
const encryptThumbnailWithin = (
decoded: { data: Uint8Array; width: number; height: number },
key: Uint8Array,
maxSize: number,
): { header: Uint8Array; ciphertext: Uint8Array } | undefined => {
for (const { maxDimension, quality } of THUMB_ENCODINGS) {
const thumbJpeg = generateThumbnail(decoded, maxDimension, quality);
const encrypted = encryptBlob(thumbJpeg, key);
if (encrypted.ciphertext.length <= maxSize) return encrypted;
}
return undefined;
};
// Regenerate and upload a thumbnail for each requested file. Originals are read
// through the library's content cache (`photo.original()`); the generated
// thumbnail is JPEG-encoded, encrypted under the file's own key, and registered
// with the server — the encrypt-and-upload path is unchanged. Each file is
// resolved to one outcome (fixed / skipped / failed) and a failure on one file
// never stops the others.
export const fixMissingThumbnails = async (
lib: Library,
client: Client,
fileIDs: number[],
onProgress?: ProgressCallback,
@@ -237,26 +169,20 @@ export const fixMissingThumbnails = async (
const log = onProgress ?? (() => {});
const results: ThumbnailFixResult[] = [];
const api = client.getApiClient();
const { userID } = client.whoami();
// Resolve each requested fileID to its file record and owning album by
// enumerating the library, each file taken from the first album that holds
// it. The raw `EnteFile` carries the per-file key the thumbnail is
// encrypted under, which the projected records deliberately do not.
const wanted = new Set(fileIDs);
const collections = await client.listCollections();
const fileMap = new Map<
number,
{ file: EnteFile; collectionName: string }
>();
for (const album of lib.albums.list()) {
for (const photo of album.photos.list()) {
if (!wanted.has(photo.fileID) || fileMap.has(photo.fileID))
continue;
const file = lib.getFile(album.collectionID, photo.fileID);
if (file) {
fileMap.set(photo.fileID, {
for (const col of collections) {
const files = await client.listFiles(col.id, col.key);
for (const file of files) {
if (fileIDs.includes(file.id) && !fileMap.has(file.id)) {
fileMap.set(file.id, {
file,
collectionName: album.name,
collectionName: col.name,
});
}
}
@@ -269,78 +195,33 @@ export const fixMissingThumbnails = async (
fileID,
title: "unknown",
collection: "unknown",
status: "failed",
reason: "file not found in any collection",
success: false,
error: "file not found in any collection",
});
continue;
}
const { file, collectionName } = entry;
const title = file.metadata.title;
const skipReason = reasonToSkip(file, userID);
if (skipReason) {
log(`[${collectionName}] Skipping ${title}: ${skipReason}`);
results.push({
fileID,
title,
collection: collectionName,
status: "skipped",
reason: skipReason,
});
continue;
}
const maxSize = file.thumbnail.size!;
const tmpDir = mkdtempSync(join(tmpdir(), "quak-thumb-"));
try {
const photo = lib.photos.byID({ fileID });
if (!photo) {
throw new Error("file not present in the library cache");
}
log(
`[${collectionName}] Downloading ${file.metadata.title} for thumbnail generation...`,
);
const origPath = join(tmpDir, "original");
await downloadFile(api, file, origPath);
log(
`[${collectionName}] Downloading ${title} for thumbnail generation...`,
`[${collectionName}] Generating thumbnail for ${file.metadata.title}...`,
);
const { path } = await photo.original();
const fileBytes = new Uint8Array(readFileSync(path));
if (!isJpeg(fileBytes)) {
const reason =
"unsupported image format (only baseline JPEG can be regenerated)";
log(`[${collectionName}] Skipping ${title}: ${reason}`);
results.push({
fileID,
title,
collection: collectionName,
status: "skipped",
reason,
});
continue;
}
log(`[${collectionName}] Generating thumbnail for ${title}...`);
const decoded = jpeg.decode(fileBytes, {
useTArray: true,
formatAsRGBA: true,
});
const fitting = encryptThumbnailWithin(decoded, file.key, maxSize);
if (!fitting) {
const reason = `no thumbnail encoding fits the recorded thumbnail size of ${maxSize} bytes`;
log(`[${collectionName}] Skipping ${title}: ${reason}`);
results.push({
fileID,
title,
collection: collectionName,
status: "skipped",
reason,
});
continue;
}
const { header, ciphertext } = fitting;
const fileBytes = readFileSync(origPath);
const thumbJpeg = generateThumbnail(new Uint8Array(fileBytes));
log(
`[${collectionName}] Uploading thumbnail (${ciphertext.length} bytes)...`,
`[${collectionName}] Encrypting and uploading thumbnail (${thumbJpeg.length} bytes)...`,
);
const { header, ciphertext } = encryptBlob(thumbJpeg, file.key);
const md5 = createHash("md5").update(ciphertext).digest("base64");
const { objectKey, url } = await api.getUploadURL(
ciphertext.length,
@@ -349,24 +230,28 @@ export const fixMissingThumbnails = async (
await api.putFile(url, ciphertext);
await api.updateThumbnail(file.id, objectKey, toBase64(header));
log(`[${collectionName}] Thumbnail uploaded for ${title}`);
results.push({
fileID,
title,
collection: collectionName,
status: "fixed",
});
} catch (err) {
log(
`[${collectionName}] FAILED ${title}: ${err instanceof Error ? err.message : err}`,
`[${collectionName}] Thumbnail uploaded for ${file.metadata.title}`,
);
results.push({
fileID,
title,
title: file.metadata.title,
collection: collectionName,
status: "failed",
reason: err instanceof Error ? err.message : String(err),
success: true,
});
} catch (err) {
log(
`[${collectionName}] FAILED ${file.metadata.title}: ${err instanceof Error ? err.message : err}`,
);
results.push({
fileID,
title: file.metadata.title,
collection: collectionName,
success: false,
error: err instanceof Error ? err.message : String(err),
});
} finally {
rmSync(tmpDir, { recursive: true, force: true });
}
}
+14 -313
View File
@@ -38,18 +38,14 @@
* the network. The fake records every call for assertion.
*/
import { describe, expect, it, vi } from "vitest";
import { describe, expect, it } from "vitest";
import {
ApiClient,
ApiError,
DEFAULT_DOWNLOAD_TIMEOUT_MS,
DEFAULT_REQUEST_TIMEOUT_MS,
} from "../../src/api/client.js";
import {
isRetryable,
isSafeToReplay,
type RetryOptions,
} from "../../src/retry.js";
import type { RetryOptions } from "../../src/retry.js";
// ---------------------------------------------------------------------------
// Test helpers
@@ -389,103 +385,6 @@ describe("ApiClient custom origins", () => {
});
});
describe("ApiClient request URLs", () => {
it("accepts a path with or without a leading slash", async () => {
const { fetch, calls } = recordingFetch(
jsonResponse({}),
jsonResponse({}),
jsonResponse({}),
);
const client = new ApiClient({ fetch });
await client.getJSON("health");
await client.postJSON("users/ott", {});
await client.putJSON("/files/thumbnail", {});
expect(calls.map((c) => c.url)).toEqual([
"https://api.ente.io/health",
"https://api.ente.io/users/ott",
"https://api.ente.io/files/thumbnail",
]);
});
it("accepts an apiOrigin with a trailing slash", async () => {
const { fetch, calls } = recordingFetch(jsonResponse({}));
const client = new ApiClient({
fetch,
apiOrigin: "https://my-ente.example.com/",
});
await client.getJSON("/health");
expect(calls[0]!.url).toBe("https://my-ente.example.com/health");
});
it("keeps a base path in a self-hosted apiOrigin for every request", async () => {
const body = new Uint8Array([1]);
const { fetch, calls } = recordingFetch(
jsonResponse({}),
jsonResponse({}),
jsonResponse({}),
streamResponse(body),
streamResponse(body),
);
const client = new ApiClient({
fetch,
apiOrigin: "https://example.com/ente/",
});
await client.getJSON("/collections/v2", { sinceTime: 0 });
await client.postJSON("/users/ott", {});
await client.putJSON("/files/thumbnail", {});
await client.getFileStream(99);
await client.getThumbnailStream(77);
expect(calls.map((c) => c.url)).toEqual([
"https://example.com/ente/collections/v2?sinceTime=0",
"https://example.com/ente/users/ott",
"https://example.com/ente/files/thumbnail",
"https://example.com/ente/files/download/99",
"https://example.com/ente/files/preview/77",
]);
});
it("percent-encodes query parameters and skips undefined ones", async () => {
const { fetch, calls } = recordingFetch(jsonResponse({}));
const client = new ApiClient({ fetch });
await client.getJSON("/search", {
q: "a&b=c/d é",
limit: 5,
cursor: undefined,
});
const url = new URL(calls[0]!.url);
expect(url.pathname).toBe("/search");
expect(url.search).toBe("?q=a%26b%3Dc%2Fd+%C3%A9&limit=5");
expect(url.searchParams.get("q")).toBe("a&b=c/d é");
});
it("rejects a path that carries its own query string", async () => {
const { fetch, calls } = recordingFetch();
const client = new ApiClient({ fetch });
await expect(client.getJSON("/diff?sinceTime=0")).rejects.toThrow(
/must not contain "\?"/,
);
await expect(client.postJSON("/users/ott?x=1", {})).rejects.toThrow(
/must not contain "\?"/,
);
expect(calls).toHaveLength(0);
});
it("rejects a path that carries a fragment", async () => {
const { fetch, calls } = recordingFetch();
const client = new ApiClient({ fetch });
await expect(client.getJSON("/diff#top")).rejects.toThrow(
/must not contain "\?" or "#"/,
);
expect(calls).toHaveLength(0);
});
});
describe("ApiError", () => {
it("throws ApiError on 4xx with status, code, requestID", async () => {
const { fetch } = recordingFetch(
@@ -740,33 +639,17 @@ describe("ApiClient retries", () => {
expect(policy.baseDelayMs).toBe(7);
expect(policy.maxDelayMs).toBe(11);
});
it("does not let a caller change its settings through that policy", async () => {
const { fetch, calls } = scriptedFetch(
textResponse("boom", 500),
textResponse("boom", 500),
textResponse("boom", 500),
);
const client = new ApiClient({
fetch,
retry: { ...noWait, attempts: 2 },
});
client.getRetryOptions().attempts = 3;
expect(client.getRetryOptions().attempts).toBe(2);
await expect(client.getJSON("/x")).rejects.toBeInstanceOf(ApiError);
expect(calls).toHaveLength(2);
});
});
describe("ApiClient timeouts", () => {
it("ships bounded default deadlines", () => {
// Asserted here so the README and the code cannot drift. The request
// deadline bounds a whole JSON call; the download deadline is an idle
// one, measured from the last byte that arrived.
// Asserted here so the README and the code cannot drift. Two numbers
// rather than one, because a deadline that is sane for a JSON call is
// nowhere near enough for a multi-gigabyte body, and a deadline long
// enough for that body would let a hung API call stall a backup for
// ten minutes.
expect(DEFAULT_REQUEST_TIMEOUT_MS).toBe(30_000);
expect(DEFAULT_DOWNLOAD_TIMEOUT_MS).toBe(60_000);
expect(DEFAULT_DOWNLOAD_TIMEOUT_MS).toBe(600_000);
});
it("attaches an abort signal to every request", async () => {
@@ -810,51 +693,6 @@ describe("ApiClient timeouts", () => {
expect(new Set(signals).size).toBe(3);
}, 5000);
it("gives every retrying entry point a fresh deadline per attempt", async () => {
// A refused connection is retried by every entry point, the
// non-idempotent ones included. If the deadline were created once,
// outside the retry, both attempts would carry the same signal.
const entryPoints: [
string,
() => Response,
(c: ApiClient) => unknown,
][] = [
["getJSON", () => jsonResponse({}), (c) => c.getJSON("/a")],
["postJSON", () => jsonResponse({}), (c) => c.postJSON("/b", {})],
["putJSON", () => jsonResponse({}), (c) => c.putJSON("/c", {})],
[
"putFile",
() => new Response(null, { status: 200 }),
(c) => c.putFile("https://s3.example/x", new Uint8Array([1])),
],
[
"getFileStream",
() => streamResponse(new Uint8Array([1])),
(c) => c.getFileStream(1),
],
[
"getThumbnailStream",
() => streamResponse(new Uint8Array([1])),
(c) => c.getThumbnailStream(1),
],
];
for (const [name, success, call] of entryPoints) {
const { fetch, calls } = scriptedFetch(
errnoError("ECONNREFUSED", "connect ECONNREFUSED"),
success(),
);
const client = new ApiClient({ fetch, retry: noWait });
await call(client);
expect(calls, name).toHaveLength(2);
const [first, second] = calls.map((c) => c.init?.signal);
expect(first, name).toBeInstanceOf(AbortSignal);
expect(second, name).toBeInstanceOf(AbortSignal);
expect(second, name).not.toBe(first);
}
});
it("recovers when a later attempt answers in time", async () => {
const { fetch, calls } = scriptedFetch(HANG, jsonResponse({ ok: 1 }));
const client = new ApiClient({
@@ -880,11 +718,6 @@ describe("ApiClient timeouts", () => {
// that never produces a chunk and never observes the signal, so the
// only thing that can unblock the read is quak's own enforcement of
// the deadline over the stream it hands out.
//
// The clock is faked, so the test runs under the real default
// deadline and waits for nothing.
vi.useFakeTimers();
try {
const stalling = new Response(
new ReadableStream<Uint8Array>({
pull: () => new Promise<void>(() => {}),
@@ -894,77 +727,16 @@ describe("ApiClient timeouts", () => {
const { fetch } = scriptedFetch(stalling);
const client = new ApiClient({
fetch,
downloadTimeoutMs: 20,
retry: { ...noWait, attempts: 1 },
});
const stream = await client.getFileStream(42);
let settled = false;
const result = readAll(stream).then(
(n) => n,
(e: unknown) => e,
);
void result.finally(() => {
settled = true;
});
const err: unknown = await readAll(stream).catch((e: unknown) => e);
await vi.advanceTimersByTimeAsync(DEFAULT_DOWNLOAD_TIMEOUT_MS - 1);
expect(settled).toBe(false);
await vi.advanceTimersByTimeAsync(1);
const err = await result;
expect(err).toBeInstanceOf(Error);
expect((err as Error).name).toBe("TimeoutError");
// Classified as every deadline is: retried by the idempotent
// downloads, never replayed for a POST or PUT.
expect(isRetryable(err)).toBe(true);
expect(isSafeToReplay(err)).toBe(false);
} finally {
vi.useRealTimers();
}
});
it("does not abort a slow body that keeps making progress", async () => {
// A deadline over the whole transfer would cut off a large video on a
// slow link however steadily it was arriving. The deadline restarts
// with every chunk, so a body that sends one byte every 600 ms for
// well over the 1000 ms deadline completes.
vi.useFakeTimers();
try {
let sent = 0;
const trickling = new Response(
new ReadableStream<Uint8Array>({
async pull(controller) {
await new Promise((resolve) =>
setTimeout(resolve, 600),
);
if (sent === 10) {
controller.close();
return;
}
controller.enqueue(new Uint8Array([sent++]));
},
}),
{ status: 200 },
);
const { fetch } = scriptedFetch(trickling);
const client = new ApiClient({
fetch,
downloadTimeoutMs: 1000,
retry: { ...noWait, attempts: 1 },
});
const stream = await client.getFileStream(42);
const result = readAll(stream).then(
(n) => n,
(e: unknown) => e,
);
await vi.advanceTimersByTimeAsync(11 * 600);
expect(await result).toBe(10);
} finally {
vi.useRealTimers();
}
});
}, 5000);
it("lets a body that arrives in time through untouched", async () => {
// The counterpart to the previous test: enforcing the deadline over
@@ -990,52 +762,6 @@ describe("ApiClient timeouts", () => {
}
expect(joined).toEqual(payload);
});
it("leaves no timer pending after a download completes or fails", async () => {
vi.useFakeTimers();
try {
const { fetch } = scriptedFetch(
streamResponse(new Uint8Array([1, 2, 3])),
textResponse("gone", 404),
);
const client = new ApiClient({
fetch,
retry: { ...noWait, attempts: 1 },
});
expect(await readAll(await client.getFileStream(1))).toBe(3);
expect(vi.getTimerCount()).toBe(0);
await expect(client.getFileStream(2)).rejects.toBeInstanceOf(
ApiError,
);
expect(vi.getTimerCount()).toBe(0);
} finally {
vi.useRealTimers();
}
});
it("never lets the download timer keep the process alive", async () => {
const spy = vi.spyOn(globalThis, "setTimeout");
try {
const { fetch } = scriptedFetch(
streamResponse(new Uint8Array([1])),
);
const client = new ApiClient({
fetch,
downloadTimeoutMs: 12_345,
retry: noWait,
});
const stream = await client.getFileStream(1);
const i = spy.mock.calls.findIndex((call) => call[1] === 12_345);
const timer = spy.mock.results[i]!.value as NodeJS.Timeout;
expect(timer.hasRef()).toBe(false);
await stream.cancel();
} finally {
spy.mockRestore();
}
});
});
describe("ApiClient error typing", () => {
@@ -1094,8 +820,9 @@ describe("ApiClient error typing", () => {
describe("ApiClient non-idempotent requests", () => {
/**
* `postJSON` and `putJSON` carry quak's requests that can change server
* state; the README lists them under "Endpoints used".
* `postJSON` and `putJSON` carry quak's only requests that change server
* state: `/users/srp/create-session`, `/users/two-factor/verify` — which
* consumes one of a small number of 2FA attempts — and `/files/thumbnail`.
*
* They are retried only on a failure that establishes no TCP connection to
* the server ever existed — DNS produced no address, or the peer refused
@@ -1195,30 +922,4 @@ describe("ApiClient non-idempotent requests", () => {
await refusedClient.updateThumbnail(1, "key", "header");
expect(refused.calls).toHaveLength(2);
});
it("does not follow or replay a redirect on POST or PUT", async () => {
// The origin has already received a request it answers with a
// redirect, so following it would let a refused connection to the
// redirect target pass for a request that never went out.
for (const send of [
(c: ApiClient) => c.postJSON("/users/ott", {}),
(c: ApiClient) => c.putJSON("/files/thumbnail", {}),
]) {
const { fetch, calls } = scriptedFetch(
new Response(null, {
status: 307,
headers: { location: "https://elsewhere.example/" },
}),
jsonResponse({}),
);
const client = new ApiClient({ fetch, retry: noWait });
const err: unknown = await send(client).catch((e: unknown) => e);
expect(calls[0]?.init?.redirect).toBe("manual");
expect(err).toBeInstanceOf(ApiError);
expect((err as ApiError).status).toBe(307);
expect(calls).toHaveLength(1);
}
});
});
+387 -974
View File
File diff suppressed because it is too large Load Diff
-732
View File
@@ -1,732 +0,0 @@
/**
* Tests for the CLI commands (`src/cli-commands.ts`, issue #12).
*
* Each command is called directly with a context whose output streams collect
* text, whose session directory is a fresh temp directory, and whose session
* loader hands back a fake client. The fake serves two albums and three files
* from memory, writes stand-in bytes for originals and thumbnails, and makes no
* network calls. The helpers the commands call (`cli-read`, `cli-output`,
* backup, thumbnails) have their own tests; these check what each command
* prints and the exit code it returns.
*/
import {
existsSync,
mkdtempSync,
readdirSync,
readFileSync,
rmSync,
statSync,
writeFileSync,
} from "node:fs";
import { join } from "node:path";
import { tmpdir } from "node:os";
import { PassThrough } from "node:stream";
import {
describe,
it,
expect,
vi,
beforeAll,
beforeEach,
afterEach,
} from "vitest";
import {
type CliContext,
saveSession,
loginCommand,
whoamiCommand,
logoutCommand,
collectionsCommand,
filesCommand,
getCommand,
getThumbCommand,
backupCommand,
backupMetadataCommand,
listMissingThumbnailsCommand,
fixMissingThumbnailsCommand,
} from "../../src/cli-commands.js";
import { run } from "../../src/cli-run.js";
import { loadSession } from "../../src/cli-session.js";
import type { Client, ClientSnapshot, LoginOptions } from "../../src/client.js";
import type { ContentSource } from "../../src/library/content.js";
import type { Collection, EnteFile } from "../../src/model/types.js";
import { init, toBase64 } from "../../src/crypto/index.js";
import { defaultCacheDirectory } from "../../src/library/index.js";
const USER_ID = 42;
const collection = (
id: number,
name: string,
isShared = false,
): Collection => ({
id,
ownerID: USER_ID,
key: new Uint8Array([id]),
name,
type: "album",
updationTime: 1,
isShared,
});
const file = (id: number, collectionID: number, title: string): EnteFile => ({
id,
collectionID,
ownerID: USER_ID,
key: new Uint8Array([id & 0xff]),
metadata: {
title,
fileType: "image",
creationTime: 1000,
modificationTime: 1000,
},
file: { decryptionHeader: "aGVhZGVy" },
thumbnail: { decryptionHeader: "dGh1bWI=" },
updationTime: 1,
});
const COLLECTIONS = [collection(1, "Vacation"), collection(2, "Work", true)];
const FILES: Record<number, EnteFile[]> = {
1: [file(100, 1, "beach.jpg"), file(101, 1, "sunset.jpg")],
2: [file(200, 2, "diagram.png")],
};
// An original is 7 bytes and a thumbnail 3. `failID` makes that file's
// original fail; `emptyThumbID` makes the server report that file's
// thumbnail as empty. `withNewFile` adds new.jpg (102) to Vacation, advancing
// the collection's updationTime as the server does, and `refreshError` makes
// listing collections fail with that message.
const fakeClient = (
opts: {
failID?: number;
emptyThumbID?: number;
withNewFile?: boolean;
refreshError?: string;
} = {},
) => {
const collections = opts.withNewFile
? [{ ...COLLECTIONS[0], updationTime: 2 }, COLLECTIONS[1]]
: COLLECTIONS;
const files = opts.withNewFile
? {
...FILES,
1: [...FILES[1], { ...file(102, 1, "new.jpg"), updationTime: 2 }],
}
: FILES;
const source: ContentSource = {
original: async ({ file: f, destination }) => {
if (f.id === opts.failID) throw new Error("HTTP 500 from server");
writeFileSync(destination, Buffer.alloc(7, f.id & 0xff));
return { bytesWritten: 7 };
},
thumbnail: async ({ file: f, destination }) => {
writeFileSync(destination, Buffer.alloc(3, f.id & 0xff));
return { bytesWritten: 3 };
},
};
const fake = {
whoami: () => ({ email: "cli@example.com", userID: USER_ID }),
collectionsSince: async () => {
if (opts.refreshError) throw new Error(opts.refreshError);
return { collections, deleted: [], cursor: 1 };
},
filesSince: async (args: { collectionID: number }) => ({
files: files[args.collectionID] ?? [],
deleted: [],
cursor: 1,
}),
contentSource: () => source,
getApiClient: () => ({
// The ML data request of `backup-metadata`: no file has any.
postJSON: async () => ({ data: [] }),
getThumbnailStream: async (fileID: number) =>
new ReadableStream<Uint8Array>({
start(controller) {
if (fileID !== opts.emptyThumbID) {
controller.enqueue(new Uint8Array(3));
}
controller.close();
},
}),
}),
};
// The commands only call the methods above.
return fake as unknown as Client;
};
// Collects everything written to it.
class Output {
text = "";
write(text: string): void {
this.text += text;
}
}
let root: string;
let stdout: Output;
let stderr: Output;
const context = (client: Client | null = fakeClient()): CliContext => ({
stdout,
stderr,
sessionDir: join(root, "session"),
cacheDir: join(root, "cache"),
loadSession: () => client,
login: async () => {
throw new Error("login not expected");
},
prompt: async () => {
throw new Error("prompt not expected");
},
promptSecret: async () => {
throw new Error("prompt not expected");
},
});
beforeAll(async () => {
await init();
});
beforeEach(() => {
root = mkdtempSync(join(tmpdir(), "quak-cli-test-"));
stdout = new Output();
stderr = new Output();
});
afterEach(() => {
rmSync(root, { recursive: true, force: true });
});
describe("session file", () => {
const snapshot: ClientSnapshot = {
email: "cli@example.com",
userID: USER_ID,
token: "token",
masterKey: "a",
secretKey: "b",
publicKey: "c",
};
it("is written with mode 0600 in a directory with mode 0700", () => {
const dir = join(root, "new", "session");
saveSession(dir, snapshot);
expect(statSync(dir).mode & 0o777).toBe(0o700);
const path = join(dir, "session.json");
expect(statSync(path).mode & 0o777).toBe(0o600);
expect(JSON.parse(readFileSync(path, "utf-8"))).toEqual(snapshot);
});
it("a missing session exits 1 with 'Not logged in'", async () => {
const ctx = { ...context(), loadSession };
expect(await whoamiCommand(ctx)).toBe(1);
expect(stderr.text).toBe(
`Not logged in. Run "quak login" first.\n` +
`Session file: ${join(ctx.sessionDir, "session.json")}\n`,
);
expect(stdout.text).toBe("");
});
it("a corrupt session exits 1 and says it is corrupt", async () => {
const ctx = { ...context(), loadSession };
saveSession(ctx.sessionDir, snapshot);
expect(await collectionsCommand(ctx, {})).toBe(1);
expect(stderr.text).toContain("is corrupt");
expect(stderr.text).toContain(
`Run "quak logout" and then "quak login" to replace it.\n`,
);
expect(stdout.text).toBe("");
});
});
// The login function is a fake that hands back a client whose snapshot is
// `snapshot`; each prompt is recorded and answered with "123456".
describe("login", () => {
const snapshot: ClientSnapshot = {
email: "cli@example.com",
userID: USER_ID,
token: "token",
masterKey: "a",
secretKey: "b",
publicKey: "c",
};
const loggedIn = {
whoami: () => ({ email: "cli@example.com", userID: USER_ID }),
toJSON: () => snapshot,
} as unknown as Client;
let prompts: string[];
const loginContext = (
login: (opts: LoginOptions) => Promise<Client>,
): CliContext => ({
...context(),
login,
prompt: async (message) => {
prompts.push(message);
return "123456";
},
promptSecret: async (message) => {
prompts.push(message);
return "123456";
},
});
beforeEach(() => {
prompts = [];
vi.stubEnv("QUAK_EMAIL", "cli@example.com");
vi.stubEnv("QUAK_PASSWORD", "hunter2");
});
afterEach(() => {
vi.unstubAllEnvs();
});
it("takes the email and password from the environment without a prompt", async () => {
const calls: LoginOptions[] = [];
const ctx = loginContext(async (opts) => {
calls.push(opts);
return loggedIn;
});
expect(await loginCommand(ctx)).toBe(0);
expect(prompts).toEqual([]);
expect(calls).toHaveLength(1);
expect(calls[0]!.email).toBe("cli@example.com");
expect(calls[0]!.password).toBe("hunter2");
const path = join(ctx.sessionDir, "session.json");
expect(stderr.text).toBe(
"Authenticating...\n" +
`Logged in as cli@example.com (user ${USER_ID})\n` +
`Session saved to ${path}\n`,
);
});
it("saves the session with mode 0600 in a directory with mode 0700", async () => {
const ctx = loginContext(async () => loggedIn);
expect(await loginCommand(ctx)).toBe(0);
expect(statSync(ctx.sessionDir).mode & 0o777).toBe(0o700);
const path = join(ctx.sessionDir, "session.json");
expect(statSync(path).mode & 0o777).toBe(0o600);
expect(JSON.parse(readFileSync(path, "utf-8"))).toEqual(snapshot);
});
it("asks for the TOTP code when the account needs one", async () => {
let code: string | undefined;
const ctx = loginContext(async (opts) => {
code = await opts.totp!();
return loggedIn;
});
expect(await loginCommand(ctx)).toBe(0);
expect(prompts).toEqual(["TOTP code: "]);
expect(code).toBe("123456");
});
it("a failed login exits 1, says why and writes no session", async () => {
const ctx = loginContext(async () => {
throw new Error("HTTP 401 from server");
});
expect(await loginCommand(ctx)).toBe(1);
expect(stderr.text).toBe(
"Authenticating...\nLogin failed: HTTP 401 from server\n",
);
expect(existsSync(join(ctx.sessionDir, "session.json"))).toBe(false);
});
});
// These use a real client read from the session file, over a fake API that
// records each request and answers with `status`.
describe("logout", () => {
const snapshot: ClientSnapshot = {
email: "cli@example.com",
userID: USER_ID,
token: "saved-token",
masterKey: toBase64(new Uint8Array(32)),
secretKey: toBase64(new Uint8Array(32)),
publicKey: toBase64(new Uint8Array(32)),
};
const requests: Request[] = [];
const logoutContext = (status: number): CliContext => ({
...context(),
loadSession: (path) =>
loadSession(path, {
fetch: async (url, init) => {
requests.push(new Request(url, init));
return new Response(JSON.stringify({}), {
status,
headers: { "content-type": "application/json" },
});
},
}),
});
beforeEach(() => {
requests.length = 0;
});
it("ends the session on the server, then deletes the file", async () => {
const ctx = logoutContext(200);
saveSession(ctx.sessionDir, snapshot);
expect(await logoutCommand(ctx)).toBe(0);
expect(requests).toHaveLength(1);
expect(requests[0]!.method).toBe("POST");
expect(new URL(requests[0]!.url).pathname).toBe("/users/logout");
expect(requests[0]!.headers.get("X-Auth-Token")).toBe("saved-token");
expect(existsSync(join(ctx.sessionDir, "session.json"))).toBe(false);
expect(stderr.text).toBe(
"Session ended on the server.\n" +
"Session deleted.\n" +
`Cache directory ${ctx.cacheDir} still holds decrypted data; delete it to remove that data.\n`,
);
});
it("still deletes the file when the server call fails, and says so", async () => {
const ctx = logoutContext(500);
saveSession(ctx.sessionDir, snapshot);
expect(await logoutCommand(ctx)).toBe(1);
expect(requests).toHaveLength(1);
expect(existsSync(join(ctx.sessionDir, "session.json"))).toBe(false);
expect(stderr.text).toBe(
"Could not end the session on the server: HTTP 500\n" +
"Session deleted.\n" +
`Cache directory ${ctx.cacheDir} still holds decrypted data; delete it to remove that data.\n`,
);
});
it("names the account's default cache directory without --cache-dir", async () => {
const ctx = { ...logoutContext(200), cacheDir: undefined };
saveSession(ctx.sessionDir, snapshot);
expect(await logoutCommand(ctx)).toBe(0);
expect(stderr.text).toContain(
`Cache directory ${defaultCacheDirectory(USER_ID)} still holds decrypted data`,
);
});
it("without a session says so, calls nothing and exits 0", async () => {
expect(await logoutCommand(logoutContext(200))).toBe(0);
expect(requests).toHaveLength(0);
expect(stderr.text).toBe("No session found.\n");
});
});
describe("whoami", () => {
it("prints the account as one line of JSON", async () => {
expect(await whoamiCommand(context())).toBe(0);
expect(stdout.text).toBe(
`{"email":"cli@example.com","userID":${USER_ID}}\n`,
);
});
});
describe("collections", () => {
it("prints one tab-separated line per album", async () => {
expect(await collectionsCommand(context(), {})).toBe(0);
expect(stdout.text).toBe(
"1\talbum\tVacation\n" + "2\talbum\tWork (shared)\n",
);
});
it("prints a JSON array with --json", async () => {
expect(await collectionsCommand(context(), { json: true })).toBe(0);
expect(JSON.parse(stdout.text)).toEqual([
{
id: 1,
name: "Vacation",
type: "album",
ownerID: USER_ID,
isShared: false,
updationTime: 1,
},
{
id: 2,
name: "Work",
type: "album",
ownerID: USER_ID,
isShared: true,
updationTime: 1,
},
]);
});
});
describe("files", () => {
it("prints one tab-separated line per file", async () => {
expect(await filesCommand(context(), { collection: "1" })).toBe(0);
expect(stdout.text).toBe(
"100\timage\tbeach.jpg\n" + "101\timage\tsunset.jpg\n",
);
});
it("prints a JSON array with --json", async () => {
const code = await filesCommand(context(), {
collection: "2",
json: true,
});
expect(code).toBe(0);
expect(JSON.parse(stdout.text)).toEqual([
{
id: 200,
title: "diagram.png",
fileType: "image",
creationTime: 1000,
collectionID: 2,
},
]);
});
it("exits 1 for an unknown collection", async () => {
expect(await filesCommand(context(), { collection: "9" })).toBe(1);
expect(stderr.text).toBe("Collection 9 not found\n");
});
it("exits 1 for a collection ID that is not a number", async () => {
expect(await filesCommand(context(), { collection: "abc" })).toBe(1);
expect(stderr.text).toBe("Invalid collection ID\n");
});
});
describe("get and get-thumb", () => {
it("get finds a file in any album without --collection", async () => {
const out = join(root, "diagram.png");
expect(await getCommand(context(), "200", { out })).toBe(0);
expect(readFileSync(out)).toEqual(Buffer.alloc(7, 200));
expect(stderr.text).toBe(`7 bytes -> ${out}\n`);
});
it("get-thumb finds a file in any album without --collection", async () => {
const out = join(root, "thumb.jpg");
expect(await getThumbCommand(context(), "200", { out })).toBe(0);
expect(readFileSync(out)).toEqual(Buffer.alloc(3, 200));
expect(stderr.text).toBe(`3 bytes -> ${out}\n`);
});
it("get exits 1 when no album has the file", async () => {
const out = join(root, "x");
expect(await getCommand(context(), "999", { out })).toBe(1);
expect(stderr.text).toBe("File 999 not found\n");
expect(existsSync(out)).toBe(false);
});
it("get-thumb exits 1 when no album has the file", async () => {
const out = join(root, "x");
expect(await getThumbCommand(context(), "999", { out })).toBe(1);
expect(stderr.text).toBe("File 999 not found\n");
expect(existsSync(out)).toBe(false);
});
it("both exit 1 for a file ID that is not a number", async () => {
expect(await getCommand(context(), "abc", {})).toBe(1);
expect(await getThumbCommand(context(), "abc", {})).toBe(1);
expect(stderr.text).toBe("Invalid file ID\nInvalid file ID\n");
});
});
describe("backup", () => {
it("exits 0 and prints a summary when every file is saved", async () => {
const dir = join(root, "backup");
expect(await backupCommand(context(), dir, {})).toBe(0);
expect(stderr.text).toContain(
"\n--- Backup complete ---\n" +
" Total files: 3\n" +
" Downloaded: 3\n" +
" Skipped: 0\n" +
" Failed: 0\n",
);
expect(stdout.text).toBe("");
});
// The backup opens its library with the precache off: it fetches the
// originals it needs into the backup, and must not also fetch every
// thumbnail in the account, or keep originals, in the per-user cache.
it("leaves nothing in the cache's originals and thumbnails", async () => {
const dir = join(root, "backup");
expect(await backupCommand(context(), dir, {})).toBe(0);
expect(readdirSync(join(root, "cache", "thumbnails"))).toEqual([]);
expect(readdirSync(join(root, "cache", "originals"))).toEqual([]);
});
it("exits 1 and lists the file when one download fails", async () => {
const ctx = context(fakeClient({ failID: 101 }));
expect(await backupCommand(ctx, join(root, "backup"), {})).toBe(1);
expect(stderr.text).toContain(" Failed: 1\n");
expect(stderr.text).toContain(
"\nFailed files:\n" +
" [Vacation] sunset.jpg (id 101): HTTP 500 from server\n",
);
});
it("prints the result as JSON with --json, still exiting 1 on a failure", async () => {
const ctx = context(fakeClient({ failID: 101 }));
const code = await backupCommand(ctx, join(root, "backup"), {
json: true,
});
expect(code).toBe(1);
const result = JSON.parse(stdout.text);
expect(result).toMatchObject({
totalFiles: 3,
downloaded: 2,
skipped: 0,
failed: 1,
});
expect(result.errors[0].fileID).toBe(101);
expect(stderr.text).toBe("Starting backup...\n");
});
it("exits 1 with the error on one line when the refresh fails", async () => {
const client = {
...fakeClient(),
collectionsSince: async () => {
throw new Error("HTTP 401 from server");
},
} as unknown as Client;
const dir = join(root, "backup");
// Through `run`, as `bin/quak.ts` does, which prints a thrown error.
const runStderr = new PassThrough();
let runText = "";
runStderr.on("data", (chunk: Buffer) => {
runText += chunk.toString();
});
const code = await new Promise<number>((resolve) => {
void run(
backupCommand(context(client), dir, {}),
new PassThrough(),
runStderr,
resolve,
);
});
expect(code).toBe(1);
expect(runText).toBe("quak: HTTP 401 from server\n");
expect(stderr.text).toBe("Starting backup...\nRefreshing library...\n");
expect(existsSync(join(dir, "originals"))).toBe(false);
});
});
describe("helper list-missing-thumbnails", () => {
it("prints one line per file with an empty thumbnail", async () => {
const ctx = context(fakeClient({ emptyThumbID: 200 }));
expect(await listMissingThumbnailsCommand(ctx, {})).toBe(0);
expect(stdout.text).toBe(
"200\tdiagram.png\tWork\tempty thumbnail (0 bytes)\n",
);
expect(stderr.text).toContain("\n1 file(s) with missing thumbnails:\n");
});
it("says so when nothing is missing", async () => {
expect(await listMissingThumbnailsCommand(context(), {})).toBe(0);
expect(stdout.text).toBe("");
expect(stderr.text).toContain("No missing thumbnails found.\n");
});
it("prints a JSON array with --json and no progress", async () => {
const ctx = context(fakeClient({ emptyThumbID: 200 }));
expect(await listMissingThumbnailsCommand(ctx, { json: true })).toBe(0);
expect(JSON.parse(stdout.text)).toEqual([
{
fileID: 200,
title: "diagram.png",
collection: "Work",
reason: "empty thumbnail (0 bytes)",
},
]);
expect(stderr.text).toBe("");
});
});
describe("backup-metadata --exif", () => {
// Runs the command and returns what it printed to stderr.
const backupMetadata = async (opts: { exif?: boolean; all?: boolean }) => {
expect(
await backupMetadataCommand(context(), join(root, "dump"), opts),
).toBe(0);
return stderr.text;
};
it("--exif extracts EXIF", async () => {
expect(await backupMetadata({ exif: true })).toContain(
"[beach.jpg] Extracting EXIF...\n",
);
});
it("--all extracts EXIF", async () => {
expect(await backupMetadata({ all: true })).toContain(
"[beach.jpg] Extracting EXIF...\n",
);
});
it("without either flag extracts no EXIF", async () => {
expect(await backupMetadata({})).not.toContain("Extracting EXIF");
});
});
// Each test first runs `collections` so the cache holds the account as it was,
// then changes the server under it.
describe("backup-metadata and the thumbnail helpers refresh first", () => {
beforeEach(async () => {
expect(await collectionsCommand(context(), {})).toBe(0);
stdout.text = "";
stderr.text = "";
});
it("backup-metadata writes a file added since the cache was written", async () => {
const ctx = context(fakeClient({ withNewFile: true }));
const dir = join(root, "dump");
expect(await backupMetadataCommand(ctx, dir, {})).toBe(0);
expect(
existsSync(join(dir, "collections", "1-Vacation", "102.json")),
).toBe(true);
});
it("list-missing-thumbnails checks a file added since the cache was written", async () => {
const ctx = context(
fakeClient({ withNewFile: true, emptyThumbID: 102 }),
);
expect(await listMissingThumbnailsCommand(ctx, {})).toBe(0);
expect(stdout.text).toBe(
"102\tnew.jpg\tVacation\tempty thumbnail (0 bytes)\n",
);
});
it("fix-missing-thumbnails finds a file added since the cache was written", async () => {
const ctx = context(fakeClient({ withNewFile: true }));
expect(
await fixMissingThumbnailsCommand(ctx, {
file: ["102"],
json: true,
}),
).toBe(0);
// Found, then skipped because the server records no thumbnail size
// for it; a file missing from the cache would fail as not found.
expect(JSON.parse(stdout.text)).toMatchObject([
{ fileID: 102, title: "new.jpg", status: "skipped" },
]);
});
// `run` in `cli-run.ts` prints a thrown error as one line and exits 1.
it("all three throw when the refresh fails", async () => {
const ctx = context(
fakeClient({ refreshError: "HTTP 503 from server" }),
);
const dir = join(root, "dump");
await expect(backupMetadataCommand(ctx, dir, {})).rejects.toThrow(
"HTTP 503 from server",
);
expect(existsSync(dir)).toBe(false);
await expect(listMissingThumbnailsCommand(ctx, {})).rejects.toThrow(
"HTTP 503 from server",
);
await expect(
fixMissingThumbnailsCommand(ctx, { file: ["100"] }),
).rejects.toThrow("HTTP 503 from server");
expect(stdout.text).toBe("");
});
});
+66 -152
View File
@@ -2,17 +2,13 @@
* Tests for `quak backup-metadata <dir>`.
*
* This command dumps all decrypted account metadata into a directory
* tree of plain JSON files, without downloading any file content (unless
* `--exif` is given). It is fast and produces a complete plaintext record of
* every collection name, file title, creation date, GPS coordinate, camera
* model, caption, face label, and any other metadata the Ente clients have
* attached.
* tree of plain JSON files, without downloading any file content. It
* is fast (no multi-megabyte downloads) and produces a complete
* plaintext record of every collection name, file title, creation
* date, GPS coordinate, camera model, caption, face label, and any
* other metadata the Ente clients have attached.
*
* As of issue #52 it runs on the library API: `runMetadataBackup(lib, client,
* dir)` enumerates collections and files from the library's cache rather than
* scanning the client directly, and `--exif` reads each original through the
* library's content cache (`photo.original()`). The ML fetch is unchanged. The
* output tree is identical:
* Layout:
*
* <dir>/
* account.json { email, userID }
@@ -38,7 +34,7 @@ import { join } from "node:path";
import { tmpdir } from "node:os";
import sodium from "libsodium-wrappers-sumo";
import { SRP, SrpServer } from "fast-srp-hap";
import { beforeAll, afterAll, describe, expect, it, vi } from "vitest";
import { beforeAll, afterAll, describe, expect, it } from "vitest";
import {
init,
toBase64,
@@ -48,21 +44,9 @@ import {
} from "../../src/crypto/index.js";
import * as jpegJs from "jpeg-js";
import { Client } from "../../src/client.js";
import { Library } from "../../src/library/index.js";
import {
runMetadataBackup,
type MetadataBackupOptions,
} from "../../src/metadata-backup.js";
import { backupMetadataCommand } from "../../src/cli-commands.js";
import { runMetadataBackup } from "../../src/metadata-backup.js";
import type { KeyAttributes } from "../../src/auth/types.js";
// One file per ML data request, so the two files of the mock account are
// fetched in two requests and one of them can fail on its own.
vi.mock("../../src/mldata-fetch.js", async (importOriginal) => ({
...(await importOriginal<typeof import("../../src/mldata-fetch.js")>()),
MLDATA_BATCH_SIZE: 1,
}));
const TEST_EMAIL = "metabackup@example.com";
const TEST_PASSWORD = "metapass";
const TEST_OPS = 2;
@@ -173,15 +157,14 @@ const buildMetaMock = async (): Promise<MetaMockState> => {
},
};
// Collection 2: "../Work" with no magic metadata. The server chose a name
// that tries to climb out of the backup directory.
// Collection 2: "Work" with no magic metadata
const ck2 = sodium.crypto_secretbox_keygen();
const { ciphertext: encCK2, nonce: ck2N } = encryptSecretbox(
ck2,
masterKey,
);
const { ciphertext: encCN2, nonce: cn2N } = encryptSecretbox(
new TextEncoder().encode("../Work"),
new TextEncoder().encode("Work"),
ck2,
);
const rawColl2 = {
@@ -355,8 +338,7 @@ const buildMetaMock = async (): Promise<MetaMockState> => {
};
};
// `failMLDataFor`: answer 500 to every ML data request that asks for this file.
const buildMetaFetch = (m: MetaMockState, failMLDataFor?: number) => {
const buildMetaFetch = (m: MetaMockState) => {
let srpServer: SrpServer;
return (async (
input: RequestInfo | URL,
@@ -412,8 +394,6 @@ const buildMetaFetch = (m: MetaMockState, failMLDataFor?: number) => {
}
if (path === "/files/data/fetch") {
const body = JSON.parse(init?.body as string);
if ((body.fileIDs as number[]).includes(failMLDataFor!))
return new Response("server error", { status: 500 });
const data = (body.fileIDs as number[])
.filter((id: number) => m.encryptedMLData[id])
.map((id: number) => ({
@@ -451,50 +431,16 @@ afterAll(() => {
rmSync(testDir, { recursive: true, force: true });
});
// Log in against the mock and open a library over its cache. The point commands
// open the library with the background precache off and a long refresh interval;
// the same here keeps the test deterministic (no thumbnail/original prefetch it
// did not ask for, no second refresh mid-test).
const openLib = async (client: Client): Promise<Library> =>
Library.open({
// The library client omits `fetchMLData`, matching how the CLI opens
// point commands: `runMetadataBackup` fetches ML data itself through
// the client, so the library's background backfill would only be a
// redundant second pass over the same endpoint.
client: {
whoami: () => client.whoami(),
collectionsSince: (args) => client.collectionsSince(args),
filesSince: (args) => client.filesSince(args),
contentSource: () => client.contentSource(),
},
cacheDirectory: mkdtempSync(join(testDir, "cache-")),
refreshIntervalSeconds: 3600,
precacheThumbnails: false,
precacheOriginals: false,
});
// Run one metadata backup end to end: fresh client, fresh library, then close.
const runBackup = async (
outDir: string,
opts?: MetadataBackupOptions,
): Promise<void> => {
describe("quak backup-metadata", () => {
it("writes account.json with email and userID", async () => {
const outDir = join(testDir, "full");
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildMetaFetch(mock) },
});
const lib = await openLib(client);
try {
await runMetadataBackup(lib, client, outDir, opts);
} finally {
lib.close();
}
};
describe("quak backup-metadata", () => {
it("writes account.json with email and userID", async () => {
const outDir = join(testDir, "full");
await runBackup(outDir);
await runMetadataBackup(client, outDir);
const account = JSON.parse(
readFileSync(join(outDir, "account.json"), "utf-8"),
@@ -505,11 +451,16 @@ describe("quak backup-metadata", () => {
it("creates per-collection directories with _collection.json", async () => {
const outDir = join(testDir, "collections");
await runBackup(outDir);
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildMetaFetch(mock) },
});
await runMetadataBackup(client, outDir);
const collDirs = readdirSync(join(outDir, "collections"));
// "../Work" is sanitized into one directory name.
expect(collDirs.sort()).toEqual(["10-Vacation", "20-__Work"]);
expect(collDirs.length).toBe(2);
// Find the Vacation collection dir (prefixed with ID)
const vacDir = collDirs.find((d) => d.includes("Vacation"))!;
@@ -527,7 +478,13 @@ describe("quak backup-metadata", () => {
it("decrypts collection-level pubMagicMetadata", async () => {
const outDir = join(testDir, "coll-magic");
await runBackup(outDir);
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildMetaFetch(mock) },
});
await runMetadataBackup(client, outDir);
const collDirs = readdirSync(join(outDir, "collections"));
const vacDir = collDirs.find((d) => d.includes("Vacation"))!;
@@ -544,7 +501,13 @@ describe("quak backup-metadata", () => {
it("writes per-file JSON with all three metadata layers", async () => {
const outDir = join(testDir, "file-meta");
await runBackup(outDir);
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildMetaFetch(mock) },
});
await runMetadataBackup(client, outDir);
const collDirs = readdirSync(join(outDir, "collections"));
const vacDir = collDirs.find((d) => d.includes("Vacation"))!;
@@ -563,7 +526,13 @@ describe("quak backup-metadata", () => {
it("handles files with no magic metadata gracefully", async () => {
const outDir = join(testDir, "no-magic");
await runBackup(outDir);
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildMetaFetch(mock) },
});
await runMetadataBackup(client, outDir);
const collDirs = readdirSync(join(outDir, "collections"));
const workDir = collDirs.find((d) => d.includes("Work"))!;
@@ -581,8 +550,14 @@ describe("quak backup-metadata", () => {
it("is incremental: second run does not fail", async () => {
const outDir = join(testDir, "incremental");
await runBackup(outDir);
await runBackup(outDir);
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildMetaFetch(mock) },
});
await runMetadataBackup(client, outDir);
await runMetadataBackup(client, outDir);
const account = JSON.parse(
readFileSync(join(outDir, "account.json"), "utf-8"),
@@ -592,7 +567,13 @@ describe("quak backup-metadata", () => {
it("fetches and decrypts ML data by default", async () => {
const outDir = join(testDir, "ml-data");
await runBackup(outDir);
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildMetaFetch(mock) },
});
await runMetadataBackup(client, outDir);
const collDirs = readdirSync(join(outDir, "collections"));
const vacDir = collDirs.find((d) => d.includes("Vacation"))!;
@@ -616,7 +597,13 @@ describe("quak backup-metadata", () => {
it("extracts EXIF from downloaded files when --exif is set", async () => {
const outDir = join(testDir, "exif-data");
await runBackup(outDir, { exif: true });
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildMetaFetch(mock) },
});
await runMetadataBackup(client, outDir, { exif: true });
const collDirs = readdirSync(join(outDir, "collections"));
const vacDir = collDirs.find((d) => d.includes("Vacation"))!;
@@ -632,78 +619,5 @@ describe("quak backup-metadata", () => {
expect(fileMeta.imageMetadata.format).toBe("jpeg");
expect(fileMeta.imageMetadata.width).toBe(100);
expect(fileMeta.imageMetadata.height).toBe(80);
expect(fileMeta.imageMetadataError).toBeUndefined();
// File 200 has no original on the mock server, so extraction fails
// and the reason is recorded instead of the field being left out.
const workDir = collDirs.find((d) => d.includes("Work"))!;
const failedMeta = JSON.parse(
readFileSync(
join(outDir, "collections", workDir, "200.json"),
"utf-8",
),
);
expect(failedMeta.imageMetadata).toBeUndefined();
expect(failedMeta.imageMetadataError).toEqual(expect.any(String));
});
});
describe("quak backup-metadata when an ML data request fails", () => {
// Run the CLI command against the mock and return its exit code, stderr
// and output directory.
const runCommand = async (failMLDataFor?: number) => {
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: {
fetch: buildMetaFetch(mock, failMLDataFor),
retry: { sleep: async () => {} },
},
});
const outDir = mkdtempSync(join(testDir, "ml-fail-"));
let stderr = "";
const code = await backupMetadataCommand(
{
stdout: { write: () => true },
stderr: { write: (text: string) => (stderr += text) },
sessionDir: testDir,
cacheDir: mkdtempSync(join(testDir, "cache-")),
loadSession: () => client,
},
outDir,
{},
);
return { code, stderr, outDir };
};
it("writes every file, marks the failed batch's files, and exits 1", async () => {
const { code, stderr, outDir } = await runCommand(200);
expect(code).toBe(1);
expect(stderr).toContain("ML data request for 1 file(s) failed");
const ok = JSON.parse(
readFileSync(
join(outDir, "collections", "10-Vacation", "100.json"),
"utf-8",
),
);
expect(ok.mlData.clip.embedding).toEqual([0.5, 0.6, 0.7]);
expect(ok.mlDataError).toBeUndefined();
const failed = JSON.parse(
readFileSync(
join(outDir, "collections", "20-__Work", "200.json"),
"utf-8",
),
);
expect(failed.metadata.title).toBe("diagram.png");
expect(failed.mlData).toBeUndefined();
expect(failed.mlDataError).toContain("500");
});
it("exits 0 when every ML data request succeeds", async () => {
const { code } = await runCommand();
expect(code).toBe(0);
});
});
-136
View File
@@ -1,136 +0,0 @@
/**
* Tests for the JPEG EXIF scan behind `quak backup-metadata --exif`.
*
* The originals come from users' libraries, so a truncated or corrupt JPEG
* must neither hang the scan nor throw out of it, and a malformed file must be
* told apart from one that simply has no EXIF: the record carries the reason in
* `exifError`. Each input below is a short hand-built byte array.
*/
import { describe, expect, it } from "vitest";
import {
extractExifFromJpeg,
extractImageMetadata,
} from "../../src/metadata-backup.js";
const SOI = [0xff, 0xd8]; // start of image
const SOS = [0xff, 0xda, 0x00, 0x02]; // start of scan, where the scan stops
const EXIF_HEADER = [0x45, 0x78, 0x69, 0x66, 0x00, 0x00]; // "Exif\0\0"
// A big-endian TIFF block with one IFD entry: Orientation (0x0112), SHORT, 6.
const TIFF_ORIENTATION_6 = [
0x4d, 0x4d, 0x00, 0x2a, 0x00, 0x00, 0x00, 0x08, 0x00, 0x01, 0x01, 0x12,
0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x00, 0x06, 0x00, 0x00, 0x00, 0x00,
0x00, 0x00,
];
// An APP1 segment whose length field matches its data.
const app1 = (data: number[]): number[] => {
const len = data.length + 2;
return [0xff, 0xe1, len >> 8, len & 0xff, ...data];
};
const bytes = (...parts: number[][]): Uint8Array =>
new Uint8Array(parts.flat());
describe("extractExifFromJpeg", () => {
it("returns the EXIF segment of a valid JPEG", () => {
const data = [...EXIF_HEADER, ...TIFF_ORIENTATION_6];
const scan = extractExifFromJpeg(bytes(SOI, app1(data), SOS));
expect(scan.error).toBeUndefined();
expect([...scan.exif!]).toEqual(data);
});
it("returns nothing for a file that is not a JPEG", () => {
const png = bytes([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a]);
expect(extractExifFromJpeg(png)).toEqual({});
});
it("returns nothing for a JPEG without EXIF", () => {
const app0 = [0xff, 0xe0, 0x00, 0x04, 0x00, 0x00];
expect(extractExifFromJpeg(bytes(SOI, app0, SOS))).toEqual({});
});
it("ignores an APP1 segment too short to hold the Exif header", () => {
// A length under 8 cannot hold the six-byte "Exif\0\0" header, so the
// segment is not EXIF. This one has length 7 and holds only "Exif\0",
// which the old code, lacking the length check, returned as EXIF.
const short = app1(EXIF_HEADER.slice(0, 5));
expect(extractExifFromJpeg(bytes(SOI, short, SOS))).toEqual({});
});
it("accepts an APP1 segment of length 8 holding just the Exif header", () => {
const scan = extractExifFromJpeg(bytes(SOI, app1(EXIF_HEADER), SOS));
expect(scan.error).toBeUndefined();
expect([...scan.exif!]).toEqual(EXIF_HEADER);
});
it("reports a JPEG truncated inside a segment header", () => {
const scan = extractExifFromJpeg(bytes(SOI, [0xff, 0xe1, 0x00]));
expect(scan.exif).toBeUndefined();
expect(scan.error).toMatch(/truncated segment length/);
});
it("reports a JPEG that ends before the image data", () => {
const app0 = [0xff, 0xe0, 0x00, 0x04, 0x00, 0x00];
const scan = extractExifFromJpeg(bytes(SOI, app0));
expect(scan.error).toMatch(/ends before the image data/);
});
it("stops on a zero-length segment instead of looping", () => {
// A length of 0 would otherwise step the scan by 2 bytes at a time
// through the rest of the file, reading garbage as markers.
const zero = [0xff, 0xe0, 0x00, 0x00];
const scan = extractExifFromJpeg(
bytes(SOI, zero, zero, zero, zero, SOS),
);
expect(scan.error).toMatch(/segment length 0 at byte 2 is too small/);
});
it("stops on a segment length of 1", () => {
const scan = extractExifFromJpeg(
bytes(SOI, [0xff, 0xe0, 0x00, 0x01], SOS),
);
expect(scan.error).toMatch(/segment length 1 at byte 2 is too small/);
});
it("reports a segment length that runs past the end of the file", () => {
// APP1 claims 0x4000 bytes but only the "Exif\0\0" header follows.
const scan = extractExifFromJpeg(
bytes(SOI, [0xff, 0xe1, 0x40, 0x00], EXIF_HEADER),
);
expect(scan.exif).toBeUndefined();
expect(scan.error).toMatch(/runs past the end of the file/);
});
});
describe("extractImageMetadata", () => {
it("parses EXIF from a valid JPEG", () => {
const meta = extractImageMetadata(
bytes(SOI, app1([...EXIF_HEADER, ...TIFF_ORIENTATION_6]), SOS),
);
expect(meta?.exifError).toBeUndefined();
expect(meta?.exif).toMatchObject({ Image: { Orientation: 6 } });
});
it("returns nothing for a file that is not a JPEG", () => {
const text = new TextEncoder().encode("just some text, not an image");
expect(extractImageMetadata(text)).toBeUndefined();
});
it("records the reason when the JPEG is malformed", () => {
const meta = extractImageMetadata(
bytes(SOI, [0xff, 0xe1, 0x40, 0x00], EXIF_HEADER),
);
expect(meta?.exif).toBeUndefined();
expect(meta?.exifError).toMatch(/runs past the end of the file/);
});
it("keeps the raw bytes and the reason when EXIF cannot be parsed", () => {
const data = [...EXIF_HEADER, 0x58, 0x58];
const meta = extractImageMetadata(bytes(SOI, app1(data), SOS));
expect(meta?.exif).toBeUndefined();
expect(meta?.exifRaw).toBe(Buffer.from(data).toString("base64"));
expect(meta?.exifError).toEqual(expect.any(String));
});
});
-94
View File
@@ -1,94 +0,0 @@
// The CLI presents a file by its own decrypted metadata, not the PhotoRecord
// projection (issue #52). For a renamed file the two disagree: the projection
// prefers `editedName` and reports `editedTime` in milliseconds, while the CLI
// must print the raw `metadata.title` and `metadata.creationTime` (microseconds)
// and name downloads after the raw title, byte-identical to the pre-library CLI.
//
// This locks in that contrast: the shared output helpers emit the raw values,
// and the projection of the same file emits the edited ones — so a regression
// that re-sourced the CLI from the projection would fail here.
import { describe, it, expect } from "vitest";
import {
fileListRow,
fileListLine,
originalName,
thumbnailName,
} from "../../src/cli-output.js";
import { deriveRecords } from "../../src/library/records.js";
import type { EnteFile } from "../../src/model/types.js";
// Microseconds, as Ente stores times.
const RAW_CREATION = 1700000000000000;
const EDITED_TIME = 1710000000000000;
const RAW_TITLE = "IMG_0001.HEIC";
const EDITED_NAME = "Sunset.heic";
// A file the user has renamed and re-dated: basic metadata holds the original
// title and capture time; public magic metadata holds the edits.
const renamedFile: EnteFile = {
id: 100,
collectionID: 10,
ownerID: 42,
key: new Uint8Array(),
metadata: {
title: RAW_TITLE,
fileType: "image",
creationTime: RAW_CREATION,
modificationTime: RAW_CREATION,
},
pubMagicMetadata: { editedName: EDITED_NAME, editedTime: EDITED_TIME },
file: { decryptionHeader: "" },
thumbnail: { decryptionHeader: "" },
updationTime: RAW_CREATION,
};
describe("CLI file output (issue #52)", () => {
it("emits the raw title and microsecond creationTime for --json", () => {
expect(fileListRow(renamedFile)).toEqual({
id: 100,
title: RAW_TITLE,
fileType: "image",
creationTime: RAW_CREATION,
collectionID: 10,
});
});
it("emits the raw title in the human column", () => {
expect(fileListLine(renamedFile)).toBe(`100\timage\t${RAW_TITLE}`);
});
it("names downloads after the raw title", () => {
expect(originalName(renamedFile)).toBe(RAW_TITLE);
expect(thumbnailName(renamedFile)).toBe(`thumb_${RAW_TITLE}`);
});
it("sanitizes the title when naming `quak get` downloads", () => {
// Without `--out`, the server-supplied title names the file, so it must
// not be able to point outside the working directory.
const hostile = {
...renamedFile,
metadata: { ...renamedFile.metadata, title: "../../.bashrc" },
};
expect(originalName(hostile)).toBe("__.._.bashrc");
expect(thumbnailName(hostile)).toBe("thumb___.._.bashrc");
const untitled = {
...renamedFile,
metadata: { ...renamedFile.metadata, title: "" },
};
expect(originalName(untitled)).toBe("file-100");
expect(thumbnailName(untitled)).toBe("thumb_file-100");
});
it("does not use the editedName/editedTime projection", () => {
const record = deriveRecords([], [renamedFile]).photos.get(100);
// The projection prefers the edits and reports milliseconds; the CLI
// helpers above deliberately do not.
expect(record?.title).toBe(EDITED_NAME);
expect(record?.takenAt).toBe(Math.floor(EDITED_TIME / 1000));
expect(fileListRow(renamedFile).title).not.toBe(record?.title);
expect(fileListRow(renamedFile).creationTime).not.toBe(record?.takenAt);
});
});
-195
View File
@@ -1,195 +0,0 @@
/**
* Tests for the CLI read helpers (`src/cli-read.ts`, owner amendment to
* issue #36, issue #52).
*
* The `collections`, `files`, `get`, and `get-thumb` commands must answer for
* current server state, not the local cache, so each helper forces a
* `Library.fresh()` round-trip before it reads. The stand-in library below
* serves nothing until `fresh()` has been awaited, so a helper that read
* without refreshing would come back empty and fail here.
*
* `collections` and `files` also list in the library's enumeration order
* (`listCollections`/`listFiles`) — the order the pre-library CLI printed — not
* the albums/photos projection's newest-first order. The fixtures are seeded in
* an enumeration order that a newest-first sort would rearrange, so a
* regression to the projection order would fail here too. Field values still
* come from the raw metadata via `cli-output.ts`.
*/
import { describe, it, expect } from "vitest";
import {
freshCollections,
freshFiles,
freshFile,
type FreshReadLibrary,
} from "../../src/cli-read.js";
import { fileListRow } from "../../src/cli-output.js";
import type { Photo } from "../../src/library/index.js";
import type { Collection, EnteFile } from "../../src/model/types.js";
const collection = (id: number, updationTime: number): Collection => ({
id,
ownerID: 42,
key: new Uint8Array(),
name: `album-${id}`,
type: "album",
updationTime,
isShared: false,
});
// Microseconds, as Ente stores times.
const file = (
id: number,
collectionID: number,
creationTime: number,
): EnteFile => ({
id,
collectionID,
ownerID: 42,
key: new Uint8Array(),
metadata: {
title: `file-${id}.jpg`,
fileType: "image",
creationTime,
modificationTime: creationTime,
},
file: { decryptionHeader: "" },
thumbnail: { decryptionHeader: "" },
updationTime: creationTime,
});
// A library that reveals its records only after `fresh()` has been awaited, and
// serves them in the enumeration order it was given. `photos.byID` returns a
// stand-in `Photo` carrying just the fileID the helper passes through.
class FakeLibrary implements FreshReadLibrary {
freshCalls = 0;
private refreshed = false;
constructor(
private readonly collections: Collection[],
private readonly files: EnteFile[],
) {}
async fresh(): Promise<unknown> {
this.freshCalls++;
this.refreshed = true;
return {};
}
listCollections(): Collection[] {
return this.refreshed ? this.collections : [];
}
getCollection(id: number): Collection | undefined {
return this.listCollections().find((c) => c.id === id);
}
listFiles(collectionID: number): EnteFile[] {
return this.refreshed
? this.files.filter((f) => f.collectionID === collectionID)
: [];
}
getFileByID(fileID: number): EnteFile | undefined {
if (!this.refreshed) return undefined;
return this.files.find((f) => f.id === fileID);
}
photos = {
byID: ({ fileID }: { fileID: number }): Photo | undefined => {
if (!this.refreshed) return undefined;
if (!this.files.some((f) => f.id === fileID)) return undefined;
return { fileID } as unknown as Photo;
},
};
}
describe("CLI read helpers (issue #36 amendment, issue #52)", () => {
it("freshCollections refreshes first, then lists in enumeration order", async () => {
// Enumeration order 2, 1, 3; a newest-first sort would be 3, 2, 1.
const lib = new FakeLibrary(
[collection(2, 200), collection(1, 300), collection(3, 100)],
[],
);
const rows = await freshCollections(lib);
expect(lib.freshCalls).toBe(1);
expect(rows.map((c) => c.id)).toEqual([2, 1, 3]);
// The projection's newest-first order is a different sequence, so this
// is not accidentally that order.
const newestFirst = [...rows]
.sort((a, b) => b.updationTime - a.updationTime)
.map((c) => c.id);
expect(newestFirst).toEqual([1, 2, 3]);
expect(rows.map((c) => c.id)).not.toEqual(newestFirst);
});
it("freshFiles refreshes first, lists in enumeration order, keeps raw fields", async () => {
// Enumeration order by id 10, 11, 12; creationTimes ascending, so a
// newest-first sort would reverse them.
const files = [
file(10, 1, 1_700_000_000_000_000),
file(11, 1, 1_700_000_000_000_001),
file(12, 1, 1_700_000_000_000_002),
];
const lib = new FakeLibrary([collection(1, 100)], files);
const rows = await freshFiles(lib, 1);
expect(lib.freshCalls).toBe(1);
expect(rows?.map((f) => f.id)).toEqual([10, 11, 12]);
// Field values come from raw metadata: microsecond creationTime and the
// raw title, unchanged.
expect(rows?.map(fileListRow)).toEqual([
{
id: 10,
title: "file-10.jpg",
fileType: "image",
creationTime: 1_700_000_000_000_000,
collectionID: 1,
},
{
id: 11,
title: "file-11.jpg",
fileType: "image",
creationTime: 1_700_000_000_000_001,
collectionID: 1,
},
{
id: 12,
title: "file-12.jpg",
fileType: "image",
creationTime: 1_700_000_000_000_002,
collectionID: 1,
},
]);
});
it("freshFiles returns undefined for an unknown collection", async () => {
const lib = new FakeLibrary([collection(1, 100)], []);
const rows = await freshFiles(lib, 999);
expect(lib.freshCalls).toBe(1);
expect(rows).toBeUndefined();
});
it("freshFile refreshes first, then resolves the photo and its raw record", async () => {
const f = file(10, 1, 1_700_000_000_000_000);
const lib = new FakeLibrary([collection(1, 100)], [f]);
const resolved = await freshFile(lib, 10);
expect(lib.freshCalls).toBe(1);
expect(resolved?.photo.fileID).toBe(10);
expect(resolved?.file.metadata.title).toBe("file-10.jpg");
expect(resolved?.file.metadata.creationTime).toBe(
1_700_000_000_000_000,
);
});
it("freshFile returns undefined for an unknown file", async () => {
const lib = new FakeLibrary([collection(1, 100)], []);
const resolved = await freshFile(lib, 404);
expect(lib.freshCalls).toBe(1);
expect(resolved).toBeUndefined();
});
});
-58
View File
@@ -1,58 +0,0 @@
/**
* Tests for `run` in `src/cli-run.ts`, which every CLI command goes through.
*/
import { PassThrough } from "node:stream";
import { describe, it, expect } from "vitest";
import { run } from "../../src/cli-run.js";
// A stream whose written text is kept in `text`; writes finish at once, so
// nothing is left waiting to drain.
const collector = (): { stream: PassThrough; text: () => string } => {
const stream = new PassThrough();
const chunks: string[] = [];
stream.on("data", (chunk: Buffer) => chunks.push(chunk.toString()));
return { stream, text: () => chunks.join("") };
};
const runToExit = async (
command: Promise<number>,
): Promise<{ code: number; stdout: string; stderr: string }> => {
const stdout = collector();
const stderr = collector();
const code = await new Promise<number>((resolve) => {
void run(command, stdout.stream, stderr.stream, resolve);
});
return { code, stdout: stdout.text(), stderr: stderr.text() };
};
describe("run", () => {
it("exits with the code the command returns", async () => {
const result = await runToExit(Promise.resolve(3));
expect(result).toEqual({ code: 3, stdout: "", stderr: "" });
});
it("prints a thrown error as one line without a stack trace and exits 1", async () => {
const failing = async (): Promise<number> => {
throw new Error(
"ENOTDIR: not a directory, mkdir '/dev/null/x/originals'",
);
};
const result = await runToExit(failing());
expect(result.code).toBe(1);
expect(result.stdout).toBe("");
expect(result.stderr).toBe(
"quak: ENOTDIR: not a directory, mkdir '/dev/null/x/originals'\n",
);
});
it("prints a thrown value that is not an Error", async () => {
const result = await runToExit(Promise.reject("offline"));
expect(result).toEqual({
code: 1,
stdout: "",
stderr: "quak: offline\n",
});
});
});
-197
View File
@@ -1,197 +0,0 @@
/**
* Tests for the client session lifecycle: `toJSON`, `fromJSON`, `logout`, and
* the CLI's `loadSession`, which reads the saved session file back into a
* client.
*/
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import sodium from "libsodium-wrappers-sumo";
import { afterAll, beforeAll, describe, expect, it } from "vitest";
import { init, toBase64 } from "../../src/crypto/index.js";
import { Client, type ClientSnapshot } from "../../src/client.js";
import { loadSession } from "../../src/cli-session.js";
const validSnapshot = (): ClientSnapshot => {
const kp = sodium.crypto_box_keypair();
return {
email: "user@example.com",
userID: 42,
token: "test-token",
masterKey: toBase64(sodium.crypto_secretbox_keygen()),
secretKey: toBase64(kp.privateKey),
publicKey: toBase64(kp.publicKey),
};
};
// The client's key buffers are private; the tests read them to prove that
// logout wipes them.
const keyBuffers = (client: Client): Uint8Array[] => [
client["masterKey"],
client["secretKey"],
client["publicKey"],
];
beforeAll(async () => {
await init();
});
describe("Client.toJSON", () => {
it("round-trips through fromJSON unchanged", () => {
const snapshot = validSnapshot();
expect(Client.fromJSON(snapshot).toJSON()).toEqual(snapshot);
});
it("throws instead of emitting a snapshot without a token", () => {
const client = Client.fromJSON(validSnapshot());
client.getApiClient().clearAuthToken();
expect(() => client.toJSON()).toThrow(/no auth token/);
});
});
describe("Client.fromJSON", () => {
const shortKey = toBase64(new Uint8Array(16));
it.each([
["email", undefined],
["email", 7],
["email", ""],
["token", undefined],
["token", null],
["token", ""],
["userID", undefined],
["userID", "42"],
["userID", 4.2],
["masterKey", undefined],
["masterKey", 7],
["masterKey", "not base64!"],
["masterKey", shortKey],
["secretKey", undefined],
["secretKey", "not base64!"],
["secretKey", shortKey],
["publicKey", undefined],
["publicKey", "not base64!"],
["publicKey", shortKey],
])("rejects %s = %j, naming the field", (field, value) => {
const snapshot: Record<string, unknown> = { ...validSnapshot() };
snapshot[field] = value;
expect(() => Client.fromJSON(snapshot)).toThrow(
new RegExp(`^Invalid session data: ${field} `),
);
});
it.each([null, "a string", 42])("rejects a non-object %j", (value) => {
expect(() => Client.fromJSON(value)).toThrow(
/^Invalid session data: not a JSON object/,
);
});
});
describe("Client.logout", () => {
it("zeroes the key buffers and clears the token", () => {
const client = Client.fromJSON(validSnapshot());
const api = client.getApiClient();
const keys = keyBuffers(client);
client.logout();
for (const key of keys) {
expect(key.length).toBe(32);
expect(key.every((b) => b === 0)).toBe(true);
}
expect(api.getAuthToken()).toBeUndefined();
});
it("makes every later operation throw", async () => {
const client = Client.fromJSON(validSnapshot());
client.logout();
expect(() => client.whoami()).toThrow(/logged out/);
expect(() => client.toJSON()).toThrow(/logged out/);
expect(() => client.getApiClient()).toThrow(/logged out/);
expect(() => client.contentSource()).toThrow(/logged out/);
await expect(client.listCollections()).rejects.toThrow(/logged out/);
await expect(client.collectionsSince({ sinceTime: 0 })).rejects.toThrow(
/logged out/,
);
await expect(
client.filesSince({
collectionID: 1,
collectionKey: new Uint8Array(32),
sinceTime: 0,
}),
).rejects.toThrow(/logged out/);
await expect(
client.fetchMLData({ fileIDs: [1], fileKeys: new Map() }),
).rejects.toThrow(/logged out/);
});
it("stops a listing in flight from decrypting with the zeroed keys", async () => {
// The server answers only after the client has logged out. If the
// listing went on to decrypt this row with all-zero keys it would fail
// with a decryption error, not the logged-out one.
const row = {
id: 1,
owner: { id: 42 },
encryptedKey: toBase64(new Uint8Array(48)),
keyDecryptionNonce: toBase64(new Uint8Array(24)),
updationTime: 1,
};
const client: Client = Client.fromJSON(validSnapshot(), {
fetch: async () => {
client.logout();
return new Response(JSON.stringify({ collections: [row] }), {
status: 200,
headers: { "content-type": "application/json" },
});
},
});
await expect(client.listCollections()).rejects.toThrow(/logged out/);
});
});
describe("loadSession", () => {
let dir: string;
beforeAll(() => {
dir = mkdtempSync(join(tmpdir(), "quak-session-test-"));
});
afterAll(() => {
rmSync(dir, { recursive: true, force: true });
});
it("returns null when there is no session file", () => {
expect(loadSession(join(dir, "missing.json"))).toBeNull();
});
it("restores a client from a valid session file", () => {
const path = join(dir, "valid.json");
writeFileSync(path, JSON.stringify(validSnapshot()));
expect(loadSession(path)!.whoami()).toEqual({
email: "user@example.com",
userID: 42,
});
});
it("says the file is corrupt when it is not JSON", () => {
const path = join(dir, "truncated.json");
writeFileSync(path, '{"email": "user@exa');
expect(() => loadSession(path)).toThrow(
`Session file ${path} is corrupt`,
);
});
it("says the file is corrupt and names the bad field", () => {
const path = join(dir, "bad-key.json");
writeFileSync(
path,
JSON.stringify({ ...validSnapshot(), secretKey: "AAAA" }),
);
expect(() => loadSession(path)).toThrow(
new RegExp(`^Session file ${path} is corrupt: .*secretKey`),
);
});
});
-33
View File
@@ -1,33 +0,0 @@
import { beforeAll, describe, expect, it } from "vitest";
import {
chunkHashFinal,
chunkHashInit,
chunkHashUpdate,
init,
} from "../../src/crypto/index.js";
beforeAll(async () => {
await init();
});
describe("content hash", () => {
// RFC 7693 Appendix A: BLAKE2b-512 of "abc".
const abc = Buffer.from(
"ba80a53f981c4d0d6a2797b69f12f6e94c212f14685ac4b74b12bb6fdbffa2d1" +
"7d87c5392aab792dc252d5de4533cc9518d38aa8dbf1925ab92386edd4009923",
"hex",
).toString("base64");
it("is unkeyed BLAKE2b-512 in standard base64", () => {
const state = chunkHashInit();
chunkHashUpdate(state, new TextEncoder().encode("abc"));
expect(chunkHashFinal(state)).toBe(abc);
});
it("gives the same hash when the input arrives in chunks", () => {
const state = chunkHashInit();
chunkHashUpdate(state, new TextEncoder().encode("a"));
chunkHashUpdate(state, new TextEncoder().encode("bc"));
expect(chunkHashFinal(state)).toBe(abc);
});
});
+4 -4
View File
@@ -29,10 +29,10 @@ describe("crypto.deriveKEK (Argon2id)", () => {
});
/**
* Cheap parameters used so the test suite stays under the 90-second
* `timeout` in the `test` phase of the `Dockerfile`. The real production
* parameters Ente uses are larger (memLimit up to 1 GiB, opsLimit 3-16).
* The algorithm is the same regardless of parameters.
* Cheap parameters used so the test suite stays under the 30-second
* budget. The real production parameters Ente uses are larger
* (memLimit up to 1 GiB, opsLimit 3-16). The algorithm is the same
* regardless of parameters.
*/
const TEST_OPS = 2;
const TEST_MEM = 64 * 1024 * 1024; // 64 MiB
+13 -398
View File
@@ -48,9 +48,7 @@
*/
import {
chmodSync,
existsSync,
mkdirSync,
readdirSync,
readFileSync,
rmSync,
@@ -61,7 +59,6 @@ import { dirname, join } from "node:path";
import { tmpdir } from "node:os";
import { createHash } from "node:crypto";
import sodium from "libsodium-wrappers-sumo";
import { zipSync } from "fflate";
import {
beforeAll,
beforeEach,
@@ -189,31 +186,7 @@ vi.mock("node:fs/promises", async (importOriginal) => {
};
});
/**
* `chunkHashUpdate` is wrapped to record the length of every piece hashed, so
* a test can show that a live photo entry reaches the hash in pieces far
* smaller than the entry, rather than decompressed whole first.
*/
const hashHook = vi.hoisted(() => ({
lengths: [] as number[],
}));
vi.mock("../../src/crypto/index.js", async (importOriginal) => {
const actual =
await importOriginal<typeof import("../../src/crypto/index.js")>();
return {
...actual,
chunkHashUpdate: (
...args: Parameters<typeof actual.chunkHashUpdate>
): void => {
hashHook.lengths.push(args[1].length);
actual.chunkHashUpdate(...args);
},
};
});
beforeEach(() => {
hashHook.lengths.length = 0;
renameHook.calls.length = 0;
renameHook.failWith = null;
durabilityHook.events.length = 0;
@@ -248,8 +221,7 @@ afterAll(() => {
* `sodium.randombytes_buf` goes through the wasm wrapper a byte at a time and
* costs roughly 20 seconds for the 4 MiB chunk below — about two hundred
* times what it costs to encrypt the same buffer, and on its own enough to
* push `make test` past the 90-second `timeout` in the `test` phase of the
* `Dockerfile`. This loop fills
* push `make test` past the 30-second cap in `script/test`. This loop fills
* 4 MiB in a few milliseconds.
*/
const patternBytes = (length: number, seed: number): Uint8Array => {
@@ -529,22 +501,6 @@ const entryPoints = [
{ name: "downloadThumbnail", download: downloadThumbnail },
];
// With no `outPath`, the destination is named after `metadata.title`, relative
// to the working directory. Such tests run inside a temporary directory:
// `make check` must not create files in the repo root.
const inDirectory = async <T>(
dir: string,
run: () => Promise<T>,
): Promise<T> => {
const previous = process.cwd();
process.chdir(dir);
try {
return await run();
} finally {
process.chdir(previous);
}
};
// ---------------------------------------------------------------------------
// Tests
// ---------------------------------------------------------------------------
@@ -580,59 +536,26 @@ describe("downloadFile", () => {
});
it("uses metadata.title as filename when outPath is omitted", async () => {
// With no `outPath`, the destination is `metadata.title`, used
// verbatim as a path. The title here is therefore given inside the
// test's temporary directory: a bare relative name would resolve
// against the process working directory, i.e. the repo root, and
// `make check` must not create files in the repo — a failure between
// the write and any cleanup would leave one behind.
const plaintext = new Uint8Array([1, 2, 3]);
const key = sodium.crypto_secretstream_xchacha20poly1305_keygen();
const { header, ciphertext } = encryptFileBody(plaintext, key);
const thumbPush =
sodium.crypto_secretstream_xchacha20poly1305_init_push(key);
const file = buildMockEnteFile(key, header, thumbPush.header);
file.metadata.title = "fallback-name.png";
const dir = mkdtempSync(join(testDir, "title-"));
const titlePath = join(testDir, "fallback-name.png");
file.metadata.title = titlePath;
const api = new ApiClient({ fetch: mockFetchForBody(ciphertext) });
const result = await inDirectory(dir, () => downloadFile(api, file));
const result = await downloadFile(api, file);
expect(result.path).toBe("fallback-name.png");
expect(readFileSync(join(dir, "fallback-name.png"))).toEqual(
Buffer.from(plaintext),
);
});
it("keeps a hostile title inside the working directory", async () => {
// The server controls the title. `../escaped.png` must not write to
// the parent directory; it becomes one file name in the current one.
const { api, file } = fixtureFor(
multiChunkKey,
multiChunk.header,
multiChunk.body,
);
file.metadata.title = "../escaped.png";
const parent = mkdtempSync(join(testDir, "hostile-"));
const dir = join(parent, "cwd");
mkdirSync(dir);
const result = await inDirectory(dir, () => downloadFile(api, file));
expect(result.path).toBe("__escaped.png");
expect(readdirSync(dir)).toEqual(["__escaped.png"]);
expect(readdirSync(parent)).toEqual(["cwd"]);
});
it("uses an explicit outPath verbatim, even one with ..", async () => {
// The caller is trusted: its path is not sanitized.
const { api, file } = fixtureFor(
multiChunkKey,
multiChunk.header,
multiChunk.body,
);
const dir = mkdtempSync(join(testDir, "explicit-"));
mkdirSync(join(dir, "sub"));
const outPath = join(dir, "sub", "..", "explicit.bin");
const result = await downloadFile(api, file, outPath);
expect(result.path).toBe(outPath);
expect(existsSync(join(dir, "explicit.bin"))).toBe(true);
expect(result.path).toBe(titlePath);
expect(readFileSync(result.path)).toEqual(Buffer.from(plaintext));
});
it("handles a larger single-chunk file (random binary payload)", async () => {
@@ -694,23 +617,6 @@ describe("downloadThumbnail", () => {
expect(result).toEqual({ path: outPath, bytesWritten: 4 });
expect(readFileSync(outPath)).toEqual(Buffer.from(plaintext));
});
it("names the thumbnail thumb_ plus the sanitized title", async () => {
const { api, file } = fixtureFor(
multiChunkKey,
multiChunk.header,
multiChunk.body,
);
file.metadata.title = "/etc/passwd";
const dir = mkdtempSync(join(testDir, "thumb-title-"));
const result = await inDirectory(dir, () =>
downloadThumbnail(api, file),
);
expect(result.path).toBe("thumb__etc_passwd");
expect(readdirSync(dir)).toEqual(["thumb__etc_passwd"]);
});
});
// ---------------------------------------------------------------------------
@@ -1016,47 +922,6 @@ describe.each(entryPoints)(
expect(readFileSync(outPath)).toEqual(Buffer.from(existing));
expect(readdirSync(dir)).toEqual(["rename-fails.bin"]);
});
it("fails without creating anything when the destination directory does not exist", async () => {
const key = sodium.crypto_secretstream_xchacha20poly1305_keygen();
const { header, ciphertext } = encryptFileBody(
patternBytes(64, 33),
key,
);
const { api, file } = fixtureFor(key, header, ciphertext);
const dir = freshDir();
const outPath = join(dir, "missing", "never.bin");
await expect(download(api, file, outPath)).rejects.toMatchObject({
code: "ENOENT",
});
// The missing directory is not created on the caller's behalf.
expect(readdirSync(dir)).toEqual([]);
});
// Root ignores directory permissions, so this fails when run as root.
// The `test` phase of the `Dockerfile` runs as the `node` user.
it("fails without creating anything when the destination directory is not writable", async () => {
const key = sodium.crypto_secretstream_xchacha20poly1305_keygen();
const { header, ciphertext } = encryptFileBody(
patternBytes(64, 34),
key,
);
const { api, file } = fixtureFor(key, header, ciphertext);
const dir = freshDir();
const outPath = join(dir, "never.bin");
chmodSync(dir, 0o500);
try {
await expect(
download(api, file, outPath),
).rejects.toMatchObject({ code: "EACCES" });
} finally {
chmodSync(dir, 0o700);
}
expect(readdirSync(dir)).toEqual([]);
});
},
);
@@ -1359,102 +1224,6 @@ describe("download retries: corruption is not retried", () => {
expect(requests()).toBe(1);
});
it("cancels the response body when decryption fails", async () => {
// A backup run carries on past a failed file, so a body left open on
// failure would hold its connection until garbage collection, once
// per failed file. This body delivers a corrupt chunk and then stays
// open, so only a cancel from the downloader can close it.
const corrupted = Uint8Array.from(multiChunk.body);
corrupted[10] ^= 0xff;
let cancelled = false;
const fetch = (async () =>
new Response(
new ReadableStream<Uint8Array>({
start(controller) {
controller.enqueue(corrupted);
},
cancel() {
cancelled = true;
},
}),
{ status: 200 },
)) as typeof globalThis.fetch;
const api = new ApiClient({ fetch, retry: { ...noWait, attempts: 1 } });
const file = buildMockEnteFile(
multiChunkKey,
multiChunk.header,
multiChunk.header,
);
const outPath = join(mkdtempSync(join(testDir, "cancel-")), "c.bin");
await expect(downloadFile(api, file, outPath)).rejects.toThrow(
/authentication failed/i,
);
expect(cancelled).toBe(true);
});
it("cancels the response body when the temp file cannot be opened", async () => {
// The download fails before a byte of the body is read, so the body
// is still open and only a cancel from the downloader can close it.
let cancelled = false;
const fetch = (async () =>
new Response(
new ReadableStream<Uint8Array>({
start(controller) {
controller.enqueue(multiChunk.body);
},
cancel() {
cancelled = true;
},
}),
{ status: 200 },
)) as typeof globalThis.fetch;
const api = new ApiClient({ fetch, retry: { ...noWait, attempts: 1 } });
const file = buildMockEnteFile(
multiChunkKey,
multiChunk.header,
multiChunk.header,
);
const outPath = join(
mkdtempSync(join(testDir, "cancel-")),
"missing",
"c.bin",
);
await expect(downloadFile(api, file, outPath)).rejects.toMatchObject({
code: "ENOENT",
});
expect(cancelled).toBe(true);
});
it("cancels the response body when the header is malformed", async () => {
// The header is rejected before the body is read, so the body is
// still open and only a cancel from the downloader can close it.
let cancelled = false;
const fetch = (async () =>
new Response(
new ReadableStream<Uint8Array>({
start(controller) {
controller.enqueue(multiChunk.body);
},
cancel() {
cancelled = true;
},
}),
{ status: 200 },
)) as typeof globalThis.fetch;
const api = new ApiClient({ fetch, retry: { ...noWait, attempts: 1 } });
const shortHeader = multiChunk.header.subarray(0, 5);
const file = buildMockEnteFile(multiChunkKey, shortHeader, shortHeader);
const outPath = join(mkdtempSync(join(testDir, "cancel-")), "c.bin");
await expect(downloadFile(api, file, outPath)).rejects.toThrow();
expect(cancelled).toBe(true);
});
});
// ---------------------------------------------------------------------------
@@ -1560,13 +1329,7 @@ describe("writeAtomic", () => {
// directory so that new entry is on disk too. Do the directory fsync
// before the rename, or skip it, and a crash can lose the rename.
expect(durabilityHook.events).toHaveLength(3);
// The temp name carries this process's ID, so a library opening the
// same cache can tell a write in progress from a leftover.
const tempSync = durabilityHook.events[0]!;
expect(tempSync.startsWith(`sync:w:${dir}/`)).toBe(true);
expect(tempSync.slice(`sync:w:${dir}/`.length)).toMatch(
new RegExp(`^\\.quak-${process.pid}-[0-9a-f]{32}\\.tmp$`),
);
expect(durabilityHook.events[0]).toMatch(/^sync:w:.*\.tmp$/);
expect(durabilityHook.events[1]).toBe(`rename:${dest}`);
expect(durabilityHook.events[2]).toBe(`sync:r:${dir}`);
});
@@ -1641,151 +1404,3 @@ describe.each(entryPoints)("$name progress", ({ name, download }) => {
expectSameBytes(readFileSync(outPath), plaintext);
});
});
describe("downloadFile content hash", () => {
// Node's own BLAKE2b-512 is the reference, so these tests do not depend
// on the code under test to compute what they expect.
const blake2b = (bytes: Uint8Array): string =>
createHash("blake2b512").update(bytes).digest("base64");
// Serve `plaintext` encrypted as file 999 with the given metadata. Four
// responses are scripted so a retried mismatch would show in `requests`.
const setup = (plaintext: Uint8Array, metadata: Partial<FileMetadata>) => {
const key = sodium.crypto_secretstream_xchacha20poly1305_keygen();
const { header, ciphertext } = encryptFileBody(plaintext, key);
const file = buildMockEnteFile(key, header, header);
file.metadata = { ...file.metadata, ...metadata };
const body = { kind: "body", bytes: ciphertext } as const;
const { fetch, requests } = scriptedCdnFetch(body, body, body, body);
const api = new ApiClient({ fetch, retry: { ...noWait, attempts: 4 } });
const dir = mkdtempSync(join(testDir, "hash-"));
const outPath = join(dir, "f.bin");
return {
run: () => downloadFile(api, file, outPath),
dir,
outPath,
requests,
};
};
const livePhotoZip = zipSync({
"image.heic": patternBytes(500, 81),
"video.mov": patternBytes(900, 82),
});
const livePhotoHash = `${blake2b(patternBytes(500, 81))}:${blake2b(patternBytes(900, 82))}`;
it("stores a file whose hash matches", async () => {
const plaintext = patternBytes(700, 80);
const t = setup(plaintext, { hash: blake2b(plaintext) });
await t.run();
expectSameBytes(readFileSync(t.outPath), plaintext);
});
it("rejects a mismatch, stores nothing, names the file and does not retry", async () => {
const t = setup(patternBytes(700, 80), {
hash: blake2b(patternBytes(700, 79)),
});
await expect(t.run()).rejects.toThrow(
/file 999: content hash .* does not match/,
);
expect(readdirSync(t.dir)).toEqual([]);
expect(t.requests()).toBe(1);
});
it("stores a file with no recorded hash unchecked", async () => {
const plaintext = patternBytes(700, 80);
const t = setup(plaintext, { hash: undefined });
await t.run();
expectSameBytes(readFileSync(t.outPath), plaintext);
});
it("stores a live photo whose image and video hashes match", async () => {
const t = setup(livePhotoZip, {
fileType: "livePhoto",
hash: livePhotoHash,
});
await t.run();
expectSameBytes(readFileSync(t.outPath), livePhotoZip);
});
it("hashes a large live photo entry as it decompresses, never whole", async () => {
// 64 MiB of zeros deflates to a few kilobytes, the shape of a ZIP
// that would exhaust memory if expanded whole.
const image = new Uint8Array(64 * 1024 * 1024);
const video = patternBytes(900, 83);
const zip = zipSync({ "image.heic": image, "video.mov": video });
const t = setup(zip, {
fileType: "livePhoto",
hash: `${blake2b(image)}:${blake2b(video)}`,
});
await t.run();
expectSameBytes(readFileSync(t.outPath), zip);
const hashed = hashHook.lengths.reduce((a, b) => a + b, 0);
expect(hashed).toBe(image.length + video.length);
expect(Math.max(...hashHook.lengths)).toBeLessThanOrEqual(
2 * STREAM_CHUNK_SIZE,
);
});
it("rejects a live photo whose hash does not match", async () => {
// The whole ZIP's hash is not the recorded one: each part is hashed.
const t = setup(livePhotoZip, {
fileType: "livePhoto",
hash: blake2b(livePhotoZip),
});
await expect(t.run()).rejects.toThrow(
/file 999: content hash .* does not match/,
);
expect(readdirSync(t.dir)).toEqual([]);
});
it("rejects a live photo that is not a readable ZIP and does not retry", async () => {
// Bytes 8-9 of a ZIP entry's local header name its compression
// method; 99 is one no reader knows, so the entry cannot be read.
const zip = livePhotoZip.slice();
zip[8] = 99;
zip[9] = 0;
const t = setup(zip, { fileType: "livePhoto", hash: livePhotoHash });
await expect(t.run()).rejects.toThrow(
/file 999: live photo is not a readable ZIP/,
);
expect(readdirSync(t.dir)).toEqual([]);
expect(t.requests()).toBe(1);
});
it("rejects a live photo ZIP with no image entry", async () => {
const zip = zipSync({ "video.mov": patternBytes(900, 82) });
const t = setup(zip, { fileType: "livePhoto", hash: livePhotoHash });
await expect(t.run()).rejects.toThrow(
/file 999: live photo ZIP does not hold both an image and a video/,
);
expect(readdirSync(t.dir)).toEqual([]);
});
it("rejects a live photo ZIP with no video entry", async () => {
const zip = zipSync({ "image.heic": patternBytes(500, 81) });
const t = setup(zip, { fileType: "livePhoto", hash: livePhotoHash });
await expect(t.run()).rejects.toThrow(
/file 999: live photo ZIP does not hold both an image and a video/,
);
expect(readdirSync(t.dir)).toEqual([]);
});
});
-93
View File
@@ -1,93 +0,0 @@
// File names built from server-supplied metadata.
//
// quak does not trust the server. A file's title and a collection's name are
// decrypted from data the server hands us, and a hostile server (or a
// compromised account) can set them to anything. quak uses them to name files
// on disk: `quak get` without `--out`, `downloadFile` without `outPath`, the
// backup's symlink and collection directories, and the extension of every file
// in the originals cache. Each of those goes through `sanitizeFileName` or
// `safeExtension`, so a title can only ever name one file inside the directory
// the caller chose.
//
// A path the user supplies (`--out`, `outPath`) is never sanitized: the caller
// is trusted, the server is not.
import { describe, expect, it } from "vitest";
import { safeExtension, sanitizeFileName } from "../../src/filename.js";
const FALLBACK = "file-42";
describe("sanitizeFileName", () => {
it("passes a normal title through unchanged", () => {
expect(sanitizeFileName("IMG_0001.HEIC", FALLBACK)).toBe(
"IMG_0001.HEIC",
);
expect(sanitizeFileName("Holiday 2024 (1).jpg", FALLBACK)).toBe(
"Holiday 2024 (1).jpg",
);
expect(sanitizeFileName("café.jpg", FALLBACK)).toBe("café.jpg");
});
it("cannot climb out of the directory with ../", () => {
// Without sanitizing, this would overwrite the user's SSH keys.
expect(sanitizeFileName("../../.ssh/authorized_keys", FALLBACK)).toBe(
"__.._.ssh_authorized_keys",
);
expect(sanitizeFileName("..", FALLBACK)).toBe("_");
expect(sanitizeFileName("..\\..\\x", FALLBACK)).toBe("__.._x");
});
it("cannot name an absolute path", () => {
expect(sanitizeFileName("/etc/passwd", FALLBACK)).toBe("_etc_passwd");
expect(sanitizeFileName("C:\\Windows\\x.dll", FALLBACK)).toBe(
"C__Windows_x.dll",
);
});
it("replaces embedded separators, so the name stays one file", () => {
expect(sanitizeFileName("a/b\\c.jpg", FALLBACK)).toBe("a_b_c.jpg");
});
it("replaces NUL and other control characters", () => {
// A NUL truncates the path in C code and makes Node's fs throw.
expect(sanitizeFileName("evil\0.jpg", FALLBACK)).toBe("evil_.jpg");
expect(sanitizeFileName("line\nbreak\x7f.jpg", FALLBACK)).toBe(
"line_break_.jpg",
);
});
it("does not produce a hidden file", () => {
expect(sanitizeFileName(".bashrc", FALLBACK)).toBe("_bashrc");
});
it("does not produce a Windows device name", () => {
expect(sanitizeFileName("CON", FALLBACK)).toBe("_CON");
expect(sanitizeFileName("nul.txt", FALLBACK)).toBe("_nul.txt");
expect(sanitizeFileName("LPT1", FALLBACK)).toBe("_LPT1");
// Only the exact names are reserved.
expect(sanitizeFileName("console.jpg", FALLBACK)).toBe("console.jpg");
});
it("falls back to the given name for an empty title", () => {
expect(sanitizeFileName("", FALLBACK)).toBe(FALLBACK);
});
});
describe("safeExtension", () => {
it("keeps a normal extension", () => {
expect(safeExtension("IMG_0001.HEIC")).toBe(".HEIC");
expect(safeExtension("clip.mp4")).toBe(".mp4");
});
it("uses .bin when there is no extension", () => {
expect(safeExtension("")).toBe(".bin");
expect(safeExtension("README")).toBe(".bin");
});
it("uses .bin when the extension holds anything but letters and digits", () => {
expect(safeExtension("x.j\\..\\pg")).toBe(".bin");
expect(safeExtension("x.jp g")).toBe(".bin");
expect(safeExtension("x.jpg\0")).toBe(".bin");
});
});
+3 -11
View File
@@ -99,10 +99,6 @@ describe("Library content wiring", () => {
cacheDirectory: join(root, "cache"),
contentSource: source,
refreshIntervalSeconds: 3600,
// On-demand wiring only; the background precache (#48) is covered
// in precache.test.ts and would race the exact-count assertions.
precacheThumbnails: false,
precacheOriginals: false,
});
const photo = lib.photos.byID({ fileID: 1 });
@@ -116,7 +112,7 @@ describe("Library content wiring", () => {
expect(lib.photos.byID({ fileID: 1 })!.record().thumbnailPath).toBe(
result.path,
);
await lib.close();
lib.close();
});
it("drives thumbnails.ensure through the cache", async () => {
@@ -126,10 +122,6 @@ describe("Library content wiring", () => {
cacheDirectory: join(root, "cache"),
contentSource: source,
refreshIntervalSeconds: 3600,
// On-demand wiring only; the background precache (#48) is covered
// in precache.test.ts and would race the exact-count assertions.
precacheThumbnails: false,
precacheOriginals: false,
});
const results = await lib.thumbnails.ensure({
@@ -139,7 +131,7 @@ describe("Library content wiring", () => {
expect(results).toEqual([
{ fileID: 1, path: join(root, "cache", "thumbnails", "1.jpg") },
]);
await lib.close();
lib.close();
});
it("throws from content methods when opened without a content source", async () => {
@@ -155,6 +147,6 @@ describe("Library content wiring", () => {
await expect(
lib.thumbnails.ensure({ fileIDs: [1], priority: "visible" }),
).rejects.toThrow(/content cache/i);
await lib.close();
lib.close();
});
});
+2 -39
View File
@@ -33,7 +33,6 @@ import {
mkdirSync,
statSync,
} from "node:fs";
import { spawnSync } from "node:child_process";
import { tmpdir } from "node:os";
import { join } from "node:path";
@@ -162,28 +161,15 @@ describe("ContentCache.open", () => {
expect(statSync(thumbnails).mode & 0o777).toBe(0o700);
});
it("removes temp files of an exited process, keeping those of a running one and complete content", async () => {
it("reaps orphan temp files but keeps complete content", async () => {
const originals = join(cacheDir, "originals");
const thumbnails = join(cacheDir, "thumbnails");
mkdirSync(originals, { recursive: true });
mkdirSync(thumbnails, { recursive: true });
// A child that has already exited: its process ID is not running.
const exitedPID = spawnSync(process.execPath, ["-e", ""]).pid;
const orphan = join(originals, `.quak-${exitedPID}-abc123.tmp`);
const orphanThumb = join(thumbnails, `.quak-${exitedPID}-abc456.tmp`);
// This test's own process stands in for another process still
// downloading into the same cache.
const inProgress = join(originals, `.quak-${process.pid}-def123.tmp`);
const inProgressThumb = join(
thumbnails,
`.quak-${process.pid}-def456.tmp`,
);
const orphan = join(originals, ".quak-abc123.tmp");
const complete = join(originals, "1.jpg");
const thumb = join(thumbnails, "2.jpg");
writeFileSync(orphan, "half-written");
writeFileSync(orphanThumb, "half-written");
writeFileSync(inProgress, "half-written");
writeFileSync(inProgressThumb, "half-written");
writeFileSync(complete, "whole");
writeFileSync(thumb, "whole-thumb");
@@ -191,12 +177,8 @@ describe("ContentCache.open", () => {
await cache.open();
expect(existsSync(orphan)).toBe(false);
expect(existsSync(orphanThumb)).toBe(false);
expect(existsSync(inProgress)).toBe(true);
expect(existsSync(inProgressThumb)).toBe(true);
expect(existsSync(complete)).toBe(true);
expect(existsSync(thumb)).toBe(true);
expect(cache.pathsFor(1)).toEqual({ originalPath: complete });
});
it("records already-cached files so their paths appear in pathsFor", async () => {
@@ -244,25 +226,6 @@ describe("ContentCache.original / thumbnail", () => {
expect(skips).toEqual(["skipped"]);
});
it("takes only a letters-and-digits extension from the title", async () => {
// The title comes from the server; an extension such as `.\..\x`
// must not reach the cache file name, so it becomes `.bin`.
const { cache } = buildCache({
files: [file(1, "a.jpg"), file(2, "b.\\..\\x"), file(3, "")],
});
await cache.open();
expect((await cache.original(1)).path).toBe(
join(cacheDir, "originals", "1.jpg"),
);
expect((await cache.original(2)).path).toBe(
join(cacheDir, "originals", "2.bin"),
);
expect((await cache.original(3)).path).toBe(
join(cacheDir, "originals", "3.bin"),
);
});
it("serves a file already present in the download directory without fetching", async () => {
const downloadDirectory = join(root, "backup");
mkdirSync(join(downloadDirectory, "originals"), { recursive: true });
-272
View File
@@ -1,272 +0,0 @@
/**
* Tests for fresh reads (issue #75, an owner amendment to design #36).
*
* The default read namespaces answer from RAM and never touch the network; a
* background loop keeps the local copy current. `fresh()` adds an awaited path:
* it forces a refresh, waits for it to complete and persist, and only then
* hands back the `albums`/`photos`/`timeline` namespaces, guaranteeing the
* local copy reflects a completed server round-trip. The contracts here:
*
* 1. A fresh read observes a server change that a same-instant default read
* would miss (the default path has not refreshed yet).
* 2. Concurrent fresh reads coalesce onto one in-flight refresh three
* concurrent `fresh()` calls make exactly one collections round-trip, not
* three.
* 3. A refresh that fails rejects the fresh read (currency was unavailable),
* while the default reads stay silent and keep serving the last good copy.
*
* The client is the same metadata-only mock the background-refresh tests use:
* no crypto, no network, scripted pages, and a record of each call. A long
* refresh interval keeps the background timer out of the way so each test's
* refreshes are exactly the ones it triggers.
*/
import { describe, it, expect, beforeEach, afterEach } from "vitest";
import { mkdtempSync, rmSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { Library } from "../../src/library/index.js";
import type { CollectionsPage, FilesPage } from "../../src/client.js";
import type { Collection, EnteFile } from "../../src/model/types.js";
const USER_ID = 42;
// Long enough that no background tick fires during a test; each test's
// refreshes are only the ones its own `fresh()` calls force.
const SLOW_INTERVAL = 3600;
const collection = (id: number, updationTime: number): Collection => ({
id,
ownerID: USER_ID,
key: new Uint8Array([id & 0xff]),
name: `album-${id}`,
type: "album",
updationTime,
isShared: false,
});
const file = (
id: number,
collectionID: number,
updationTime: number,
): EnteFile => ({
id,
collectionID,
ownerID: USER_ID,
key: new Uint8Array([id & 0xff]),
metadata: {
title: `file-${id}.jpg`,
fileType: "image",
creationTime: updationTime,
modificationTime: updationTime,
},
file: { decryptionHeader: "aGVhZGVy" },
thumbnail: { decryptionHeader: "dGh1bWI=" },
updationTime,
});
class MockClient {
userID = USER_ID;
failCollections = false;
collectionsQueue: CollectionsPage[] = [];
filesByCollection = new Map<number, FilesPage[]>();
collectionsSinceTimes: number[] = [];
filesCalls: { collectionID: number; sinceTime: number }[] = [];
whoami(): { email: string; userID: number } {
return { email: "user@example.com", userID: this.userID };
}
async collectionsSince(args: {
sinceTime: number;
}): Promise<CollectionsPage> {
this.collectionsSinceTimes.push(args.sinceTime);
if (this.failCollections) throw new Error("network down");
return (
this.collectionsQueue.shift() ?? {
collections: [],
deleted: [],
cursor: args.sinceTime,
}
);
}
async filesSince(args: {
collectionID: number;
collectionKey: Uint8Array;
sinceTime: number;
}): Promise<FilesPage> {
this.filesCalls.push({
collectionID: args.collectionID,
sinceTime: args.sinceTime,
});
const queue = this.filesByCollection.get(args.collectionID);
return (
queue?.shift() ?? {
files: [],
deleted: [],
cursor: args.sinceTime,
}
);
}
filesFor(collectionID: number, ...pages: FilesPage[]): void {
this.filesByCollection.set(collectionID, pages);
}
}
// A client seeded with one collection and one file, opened with an empty cache
// so the initial refresh is awaited and post-`open()` state is deterministic.
const openSeeded = async (
cacheDirectory: string,
): Promise<{ client: MockClient; lib: Library }> => {
const client = new MockClient();
client.collectionsQueue.push({
collections: [collection(1, 100)],
deleted: [],
cursor: 100,
});
client.filesFor(1, {
files: [file(1001, 1, 90)],
deleted: [],
cursor: 90,
});
const lib = await Library.open({
client,
cacheDirectory,
refreshIntervalSeconds: SLOW_INTERVAL,
});
return { client, lib };
};
describe("Library.fresh", () => {
let dir: string;
let cacheDirectory: string;
beforeEach(() => {
dir = mkdtempSync(join(tmpdir(), "quak-fresh-"));
cacheDirectory = join(dir, "cache");
});
afterEach(() => {
rmSync(dir, { recursive: true, force: true });
});
it("observes a server change a same-instant default read would miss", async () => {
const { client, lib } = await openSeeded(cacheDirectory);
try {
// A new file appears on the server after open, advancing its
// collection so the next refresh re-enumerates it.
client.collectionsQueue.push({
collections: [collection(1, 200)],
deleted: [],
cursor: 200,
});
client.filesFor(1, {
files: [file(1002, 1, 190)],
deleted: [],
cursor: 190,
});
// A default read at this instant has not refreshed: it misses 1002.
expect(lib.photos.byID({ fileID: 1002 })).toBeUndefined();
// A fresh read forces the round-trip and sees it.
const reads = await lib.fresh();
expect(reads.photos.byID({ fileID: 1002 })?.fileID).toBe(1002);
// And the change is now live for the default namespaces too.
expect(lib.photos.byID({ fileID: 1002 })?.fileID).toBe(1002);
} finally {
await lib.close();
}
});
it("coalesces concurrent fresh reads onto one in-flight refresh", async () => {
const { client, lib } = await openSeeded(cacheDirectory);
try {
client.collectionsQueue.push({
collections: [collection(1, 200)],
deleted: [],
cursor: 200,
});
client.filesFor(1, {
files: [file(1002, 1, 190)],
deleted: [],
cursor: 190,
});
const collectionsBefore = client.collectionsSinceTimes.length;
const filesBefore = client.filesCalls.length;
// Gate the next collections fetch so all three fresh reads are in
// flight together before any of them completes.
let release: () => void = () => {};
const gate = new Promise<void>((r) => {
release = r;
});
const inner = client.collectionsSince.bind(client);
client.collectionsSince = async (args: { sinceTime: number }) => {
await gate;
return inner(args);
};
const all = Promise.all([lib.fresh(), lib.fresh(), lib.fresh()]);
release();
const [a, b, c] = await all;
// Exactly one collections round-trip and one file round-trip served
// all three fresh reads.
expect(client.collectionsSinceTimes.length).toBe(
collectionsBefore + 1,
);
expect(client.filesCalls.length).toBe(filesBefore + 1);
// All three observed the change.
for (const reads of [a, b, c]) {
expect(reads.photos.byID({ fileID: 1002 })?.fileID).toBe(1002);
}
} finally {
await lib.close();
}
});
it("rejects the fresh read when the refresh fails, leaving defaults intact", async () => {
const { client, lib } = await openSeeded(cacheDirectory);
try {
client.failCollections = true;
// Two concurrent fresh reads both reject, and share one failed
// round-trip rather than each making its own.
const collectionsBefore = client.collectionsSinceTimes.length;
const first = lib.fresh();
const second = lib.fresh();
await expect(first).rejects.toThrow(/network down/);
await expect(second).rejects.toThrow(/network down/);
expect(client.collectionsSinceTimes.length).toBe(
collectionsBefore + 1,
);
// The default reads never rejected: they still serve the last good
// copy, and the failure surfaced through status().
expect(lib.photos.byID({ fileID: 1001 })?.fileID).toBe(1001);
expect(lib.status().lastError).toMatch(/network down/);
// Recovery: once the server answers, a fresh read resolves current.
client.failCollections = false;
client.collectionsQueue.push({
collections: [collection(2, 300)],
deleted: [],
cursor: 300,
});
const reads = await lib.fresh();
expect(reads.albums.byID({ collectionID: 2 })?.collectionID).toBe(
2,
);
expect(lib.status().lastError).toBeUndefined();
} finally {
await lib.close();
}
});
});
+17 -89
View File
@@ -17,8 +17,7 @@
* failure surfaces via `onProgress` ("failed") and `status()`, and a later
* success clears the error. `open()` itself resolves even when the first
* refresh fails (offline start from cache).
* 5. `close()` stops the timer and is idempotent, and its promise resolves
* only once an in-flight refresh has written the cache file.
* 5. `close()` stops the timer and is idempotent.
* 6. `cacheDirectory` defaults to the env-paths cache dir plus the user id.
* 7. `open()` branches on the cache: an empty cache awaits the first refresh
* (it has nothing to serve yet); an existing cache serves its copy at once
@@ -38,11 +37,6 @@
* short interval and `vi.waitFor`: a fake clock cannot settle the real
* fsync-and-rename cache write, and empty diffs never write, so the eventual
* state is stable to poll for.
*
* A refresh changes RAM before it writes the cache file, so a polled state can
* be visible while that write is still running. Every test therefore awaits
* `close()`, which waits for the in-flight refresh, before `afterEach` removes
* the directory.
*/
import { describe, it, expect, beforeEach, afterEach, vi } from "vitest";
@@ -199,7 +193,7 @@ describe("Library.open and background refresh", () => {
expect(reloaded.getFile(1, 1001)?.id).toBe(1001);
expect(reloaded.collectionsSinceTime).toBe(100);
} finally {
await lib.close();
lib.close();
}
});
@@ -230,7 +224,7 @@ describe("Library.open and background refresh", () => {
expect(client.collectionsSinceTimes.length).toBe(collectionCalls);
expect(client.filesCalls.length).toBe(fileCalls);
} finally {
await lib.close();
lib.close();
}
});
@@ -277,7 +271,7 @@ describe("Library.open and background refresh", () => {
{ timeout: 2000, interval: 5 },
);
} finally {
await lib.close();
lib.close();
}
});
@@ -309,7 +303,7 @@ describe("Library.open and background refresh", () => {
// never re-fetched.
expect(client.filesCalls).toEqual([]);
} finally {
await lib.close();
lib.close();
}
});
@@ -363,7 +357,7 @@ describe("Library.open and background refresh", () => {
{ timeout: 2000, interval: 5 },
);
} finally {
await lib.close();
lib.close();
}
});
@@ -408,7 +402,7 @@ describe("Library.open and background refresh", () => {
await new Promise((r) => setTimeout(r, FAST_INTERVAL * 1000 * 4));
expect(saveSpy).toHaveBeenCalledTimes(2);
} finally {
await lib.close();
lib.close();
saveSpy.mockRestore();
}
});
@@ -468,7 +462,7 @@ describe("Library.open and background refresh", () => {
{ timeout: 2000, interval: 5 },
);
} finally {
await lib.close();
lib.close();
}
});
@@ -494,7 +488,7 @@ describe("Library.open and background refresh", () => {
),
).toBe(true);
} finally {
await lib.close();
lib.close();
}
});
@@ -508,14 +502,9 @@ describe("Library.open and background refresh", () => {
seed.putFile(file(1001, 1, 400));
await seed.save();
// The server does not answer this run's first refresh until the test
// is done with it.
let answerFirstFetch: (page: CollectionsPage) => void = () => {};
// The server never answers this run's first refresh.
const client = new MockClient();
client.collectionsSince = () =>
new Promise<CollectionsPage>((resolve) => {
answerFirstFetch = resolve;
});
client.collectionsSince = () => new Promise<CollectionsPage>(() => {});
// open() must resolve from the cache without blocking on the network,
// and reads must serve the seeded copy.
@@ -528,9 +517,7 @@ describe("Library.open and background refresh", () => {
expect(lib.status().lastRefreshAt).toBeUndefined();
expect(lib.status().lastError).toBeUndefined();
} finally {
// close() waits for the outstanding refresh, so let it finish.
answerFirstFetch({ collections: [], deleted: [], cursor: 500 });
await lib.close();
lib.close();
}
});
@@ -577,7 +564,7 @@ describe("Library.open and background refresh", () => {
expect(lib.listFiles(1).map((f) => f.id)).toEqual([1001]);
expect(lib.status().lastRefreshAt).toBeGreaterThan(0);
} finally {
await lib.close();
lib.close();
}
});
@@ -632,7 +619,7 @@ describe("Library.open and background refresh", () => {
);
expect(reloaded.getFile(1, 1001)?.id).toBe(1001);
} finally {
await lib.close();
lib.close();
saveSpy.mockRestore();
}
});
@@ -647,8 +634,8 @@ describe("Library.open and background refresh", () => {
});
const callsAfterOpen = client.collectionsSinceTimes.length;
await lib.close();
await lib.close(); // second close must not throw
lib.close();
lib.close(); // second close must not throw
expect(lib.status().closed).toBe(true);
// No further refreshes fire once closed.
@@ -656,65 +643,6 @@ describe("Library.open and background refresh", () => {
expect(client.collectionsSinceTimes.length).toBe(callsAfterOpen);
});
it("close() resolves only after an in-flight refresh has written the cache", async () => {
const path = join(cacheDirectory, "metadata.json");
const seed = await MetadataStore.load(path);
seed.userID = USER_ID;
seed.collectionsSinceTime = 500;
seed.putCollection(collection(1, 400));
seed.putFile(file(1001, 1, 400));
await seed.save();
const client = new MockClient();
client.collectionsQueue.push({
collections: [collection(1, 600)],
deleted: [],
cursor: 600,
});
client.filesFor(1, {
files: [file(1002, 1, 600)],
deleted: [],
cursor: 600,
});
// Hold the refresh's cache write until the test releases it.
const realSave = MetadataStore.prototype.save;
let releaseSave: () => void = () => {};
const saveHeld = new Promise<void>((resolve) => {
releaseSave = resolve;
});
const saveSpy = vi
.spyOn(MetadataStore.prototype, "save")
.mockImplementation(async function (this: MetadataStore) {
await saveHeld;
return realSave.call(this);
});
const lib = await Library.open({ client, cacheDirectory });
try {
await vi.waitFor(() => expect(saveSpy).toHaveBeenCalled(), {
timeout: 2000,
interval: 5,
});
let closed = false;
const closing = lib.close().then(() => {
closed = true;
});
await new Promise((r) => setTimeout(r, 50));
expect(closed).toBe(false);
releaseSave();
await closing;
const reloaded = await MetadataStore.load(path);
expect(reloaded.getFile(1, 1002)?.id).toBe(1002);
} finally {
releaseSave();
await lib.close();
saveSpy.mockRestore();
}
});
it("defaults cacheDirectory to the env-paths cache dir plus user id", async () => {
const xdg = join(dir, "xdg-cache");
const prev = process.env.XDG_CACHE_HOME;
@@ -731,7 +659,7 @@ describe("Library.open and background refresh", () => {
expect(lib.cacheDirectory.startsWith(xdg)).toBe(true);
expect(lib.cacheDirectory.endsWith(String(USER_ID))).toBe(true);
} finally {
await lib.close();
lib.close();
}
} finally {
if (prev === undefined) delete process.env.XDG_CACHE_HOME;
+2 -115
View File
@@ -13,8 +13,6 @@
* 2. `Library` wiring: after each refresh the library fetches ML data through
* the metadata pool for every known file not yet cached, is incremental on
* later refreshes, and refetches a file whose `updationTime` advanced.
* Opening a cache directory written by another account starts empty,
* its ML data included (issue #104).
*
* Embedding values are chosen to be exactly representable as float32 so the
* round-trip through `clip.f32` compares equal.
@@ -376,7 +374,7 @@ describe("Library ML-data fetch on refresh", () => {
await new Promise((r) => setTimeout(r, FAST_INTERVAL * 1000 * 5));
expect(client.mlFetchCalls.length).toBe(callsAfterFirst);
} finally {
await lib.close();
lib.close();
}
});
@@ -445,118 +443,7 @@ describe("Library ML-data fetch on refresh", () => {
expect(client.mlFetchCalls.length).toBeGreaterThan(callsBefore);
expect(client.mlFetchCalls.flat()).toContain(1001);
} finally {
await lib.close();
}
});
it("close() resolves only after a running ML data fetch has stored its payloads", async () => {
const client = new MLMockClient();
client.collectionsQueue.push({
collections: [collection(1, 100)],
deleted: [],
cursor: 100,
});
client.filesFor(1, {
files: [file(1001, 1, 90)],
deleted: [],
cursor: 90,
});
client.mlByFile.set(1001, payload([0.5, 0.25, 0.75]));
// Hold the ML data fetch open until the test releases it.
let release!: () => void;
const held = new Promise<void>((r) => (release = r));
let fetchStarted!: () => void;
const started = new Promise<void>((r) => (fetchStarted = r));
const realFetch = client.fetchMLData.bind(client);
client.fetchMLData = async (args) => {
fetchStarted();
await held;
return realFetch(args);
};
const lib = await Library.open({
client,
cacheDirectory,
refreshIntervalSeconds: 3600,
});
try {
await started;
let closed = false;
const closing = lib.close().then(() => {
closed = true;
});
await new Promise((r) => setTimeout(r, 20));
expect(closed).toBe(false);
release();
await closing;
expect(
existsSync(join(cacheDirectory, "mldata", "1001.json")),
).toBe(true);
} finally {
release();
await lib.close();
}
});
it("starts empty when the cache directory holds another account's cache", async () => {
// Account A fills the cache directory: metadata and ML data.
const clientA = new MLMockClient();
clientA.collectionsQueue.push({
collections: [collection(1, 100)],
deleted: [],
cursor: 100,
});
clientA.filesFor(1, {
files: [file(1001, 1, 90)],
deleted: [],
cursor: 90,
});
clientA.mlByFile.set(1001, payload([0.5, 0.25, 0.75]));
const libA = await Library.open({
client: clientA,
cacheDirectory,
refreshIntervalSeconds: 3600,
});
try {
await vi.waitFor(
() => expect(libA.status().lastMLFetchAt).toBeGreaterThan(0),
{ timeout: 2000, interval: 5 },
);
} finally {
await libA.close();
}
// Account B opens the same directory.
const clientB = new MLMockClient();
clientB.userID = USER_ID + 1;
const sinceTimes: number[] = [];
const realCollectionsSince = clientB.collectionsSince.bind(clientB);
clientB.collectionsSince = async (args) => {
sinceTimes.push(args.sinceTime);
return realCollectionsSince(args);
};
const libB = await Library.open({
client: clientB,
cacheDirectory,
refreshIntervalSeconds: 3600,
});
try {
expect(sinceTimes[0]).toBe(0);
expect(libB.status().userID).toBe(USER_ID + 1);
expect(libB.listCollections()).toEqual([]);
expect(libB.getFile(1, 1001)).toBeUndefined();
expect(await libB.mldata.forFile({ fileID: 1001 })).toBeUndefined();
expect(
libB.mldata.searchByEmbedding({ embedding: [0.5, 0.25, 0.75] }),
).toEqual([]);
expect(
existsSync(join(cacheDirectory, "mldata", "1001.json")),
).toBe(false);
} finally {
await libB.close();
lib.close();
}
});
});
-108
View File
@@ -1,108 +0,0 @@
/**
* Tests for the content-similarity search surface over the CLIP index
* (issue #50).
*
* The surface is `lib.mldata`: `forFile` reads the full stored payload from
* disk, while `similar` and `searchByEmbedding` rank fileIDs by cosine
* similarity over the in-RAM `Float32Array` index alone (no disk, no network).
* The fixture uses axis-aligned vectors so the correct cosine ranking is
* obvious by inspection; cosine ignores magnitude, so `[2, 0, 0]` ranks above
* `[0.8, 0.6, 0]` for a `[1, 0, 0]` query.
*/
import { describe, it, expect, beforeEach, afterEach } from "vitest";
import { mkdtempSync, rmSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { MLDataStore } from "../../src/library/mldata.js";
import { makeMLDataAPI, type MLDataAPI } from "../../src/library/mlsearch.js";
import type { MLData } from "../../src/mldata-fetch.js";
// A payload shaped like Ente's: a CLIP embedding plus face data that only the
// on-disk payload carries (never the RAM index).
const payload = (embedding: number[]): MLData => ({
face: { faces: [{ faceID: "f", detection: { box: { x: 0.5 } } }] },
clip: { embedding },
});
// A small fixture index. Directions are chosen so every cosine ranking below
// is unambiguous.
const fixture = (): Map<number, MLData> =>
new Map([
[10, payload([1, 0, 0])],
[20, payload([0.8, 0.6, 0])],
[30, payload([0, 1, 0])],
[40, payload([-1, 0, 0])],
[50, payload([2, 0, 0])],
]);
describe("lib.mldata content-similarity search", () => {
let dir: string;
let store: MLDataStore;
let api: MLDataAPI;
beforeEach(async () => {
dir = mkdtempSync(join(tmpdir(), "quak-mlsearch-"));
store = await MLDataStore.open(dir);
const updation = new Map([...fixture().keys()].map((id) => [id, 1]));
await store.storeFetched(fixture(), updation);
api = makeMLDataAPI(() => store);
});
afterEach(() => {
rmSync(dir, { recursive: true, force: true });
});
it("forFile returns the whole stored payload, or undefined when uncached", async () => {
const full = await api.forFile({ fileID: 20 });
expect(full).toBeDefined();
// Face data lives only in the payload, never in the RAM index.
expect(full?.face).toBeDefined();
expect(full?.clip).toEqual({ embedding: [0.8, 0.6, 0] });
expect(await api.forFile({ fileID: 999 })).toBeUndefined();
});
it("similar ranks other files by cosine and excludes the query itself", () => {
// Query is file 10 = [1, 0, 0]. By cosine: 50 (1.0) > 20 (0.8) >
// 30 (0) > 40 (-1); 10 itself is left out.
const ranked = api.similar({ fileID: 10 });
expect(ranked.map((r) => r.fileID)).toEqual([50, 20, 30, 40]);
// Cosine ignores magnitude: [2,0,0] is a perfect match for [1,0,0].
expect(ranked[0]).toMatchObject({ fileID: 50 });
expect(ranked[0].score).toBeCloseTo(1, 5);
});
it("similar honours limit and returns [] for an unindexed file", () => {
expect(
api.similar({ fileID: 10, limit: 2 }).map((r) => r.fileID),
).toEqual([50, 20]);
expect(api.similar({ fileID: 999 })).toEqual([]);
});
it("searchByEmbedding ranks the index by cosine to the query vector", () => {
// Query [0, 1, 0]: 30 (1.0) > 20 (0.6) > {10, 40, 50} all 0, broken by
// ascending fileID.
const ranked = api.searchByEmbedding({ embedding: [0, 1, 0] });
expect(ranked.map((r) => r.fileID)).toEqual([30, 20, 10, 40, 50]);
expect(ranked[0].score).toBeCloseTo(1, 5);
expect(
api
.searchByEmbedding({ embedding: [0, 1, 0], limit: 2 })
.map((r) => r.fileID),
).toEqual([30, 20]);
});
it("searchByEmbedding returns [] for a wrong-length or zero query", () => {
expect(api.searchByEmbedding({ embedding: [1, 0] })).toEqual([]);
expect(api.searchByEmbedding({ embedding: [0, 0, 0] })).toEqual([]);
});
it("degrades to empty results when no ML store is present", async () => {
const none = makeMLDataAPI(() => undefined);
expect(await none.forFile({ fileID: 10 })).toBeUndefined();
expect(none.similar({ fileID: 10 })).toEqual([]);
expect(none.searchByEmbedding({ embedding: [1, 0, 0] })).toEqual([]);
});
});
-498
View File
@@ -1,498 +0,0 @@
/**
* The aggressive local precache (issue #48), driven from `Library.open`.
*
* Two background fills start with no caller input: every thumbnail in the
* account newest first, and the originals of the pinned set (the favorites
* album then the latest `precacheOriginalsDays` window ending at the newest
* file). Both run through the shared pools at background priority, so on-demand
* work always preempts them; both report through `status()`. The pinned set is
* the eviction predicate (#47), so a pinned original is never evicted and a
* file that leaves the set becomes an ordinary, evictable original.
*
* The unit tests drive `Precache` against a fake cache that records what it was
* asked to fetch (order and kind) with no pool or network; the integration
* tests drive the real wiring through `Library.open` with a stub content
* source, and the eviction test drives the real `ContentCache`.
*/
import { describe, it, expect, beforeEach, afterEach } from "vitest";
import { mkdtempSync, rmSync, existsSync, utimesSync } from "node:fs";
import { writeFile } from "node:fs/promises";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { Precache, type PrecacheCache } from "../../src/library/precache.js";
import {
deriveRecords,
type DerivedRecords,
} from "../../src/library/records.js";
import {
ContentCache,
type ContentSource,
type EnsureResult,
type StatFsFn,
} from "../../src/library/content.js";
import { RequestPools } from "../../src/library/pools.js";
import { Library } from "../../src/library/index.js";
import type { Collection, EnteFile } from "../../src/model/types.js";
import type { CollectionsPage, FilesPage } from "../../src/client.js";
const DAY_MICROS = 24 * 60 * 60 * 1000 * 1000;
const collection = (
id: number,
type: Collection["type"] = "album",
): Collection => ({
id,
ownerID: 1,
key: new Uint8Array([id & 0xff]),
name: `album-${id}`,
type,
updationTime: 1,
isShared: false,
});
// A file whose creationTime (microseconds) places it `daysAgo` days before a
// fixed reference instant, so the latest-week window is deterministic.
const REFERENCE_MICROS = 1_000 * DAY_MICROS;
const file = (id: number, collectionID: number, daysAgo: number): EnteFile => ({
id,
collectionID,
ownerID: 1,
key: new Uint8Array([id & 0xff]),
metadata: {
title: `file-${id}.jpg`,
fileType: "image",
creationTime: REFERENCE_MICROS - daysAgo * DAY_MICROS,
modificationTime: 0,
},
file: { decryptionHeader: "aGVhZGVy" },
thumbnail: { decryptionHeader: "dGh1bWI=" },
updationTime: 1,
});
const records = (
collections: Collection[],
files: EnteFile[],
): DerivedRecords => deriveRecords(collections, files);
// A fake cache: records every fetch (kind + order) and reports the files it has
// stored via `pathsFor`. `presentThumbs`/`presentOriginals` seed already-cached
// files so the precache skips them with a single lookup.
class FakeCache implements PrecacheCache {
readonly thumbFetched: number[] = [];
readonly originalFetched: number[] = [];
readonly presentThumbs = new Set<number>();
readonly presentOriginals = new Set<number>();
pathsFor(fileID: number): {
originalPath?: string;
thumbnailPath?: string;
} {
const out: { originalPath?: string; thumbnailPath?: string } = {};
if (this.presentThumbs.has(fileID))
out.thumbnailPath = `/thumbs/${fileID}`;
if (this.presentOriginals.has(fileID))
out.originalPath = `/originals/${fileID}`;
return out;
}
async ensureThumbnails(args: {
fileIDs: number[];
priority: "background";
}): Promise<EnsureResult[]> {
return args.fileIDs.map((fileID) => {
this.thumbFetched.push(fileID);
this.presentThumbs.add(fileID);
return { fileID, path: `/thumbs/${fileID}` };
});
}
async ensureOriginals(args: {
fileIDs: number[];
}): Promise<EnsureResult[]> {
return args.fileIDs.map((fileID) => {
this.originalFetched.push(fileID);
this.presentOriginals.add(fileID);
return { fileID, path: `/originals/${fileID}` };
});
}
}
// Resolve once a predicate holds, polling the microtask queue; fails fast
// rather than hanging the suite.
const until = async (predicate: () => boolean): Promise<void> => {
for (let i = 0; i < 1000; i++) {
if (predicate()) return;
await new Promise((r) => setTimeout(r, 1));
}
throw new Error("condition not met in time");
};
describe("Precache unit", () => {
it("precaches every thumbnail newest first, skipping present ones", async () => {
const cols = [collection(1)];
const files = [
file(1, 1, 0),
file(2, 1, 1),
file(3, 1, 2),
file(4, 1, 3),
];
const cache = new FakeCache();
cache.presentThumbs.add(3); // already on disk: skipped
const pre = new Precache({ originals: false });
pre.bind(cache);
pre.update(records(cols, files));
pre.start();
await until(() => cache.thumbFetched.length === 3);
// Newest first (file 1 is newest), file 3 skipped by a lookup.
expect(cache.thumbFetched).toEqual([1, 2, 4]);
expect(cache.originalFetched).toEqual([]);
});
it("pins favorites then the latest-week window and precaches their originals in that order", async () => {
const cols = [collection(1), collection(2, "favorites")];
// File 10 is an old favorite (30 days old); files 1..3 are within the
// 7-day window; file 4 is outside it.
const files = [
file(1, 1, 0),
file(2, 1, 2),
file(3, 1, 6),
file(4, 1, 20),
file(10, 2, 30), // favorite, old
];
const cache = new FakeCache();
const pre = new Precache({ originalsDays: 7 });
pre.bind(cache);
pre.update(records(cols, files));
pre.start();
await until(() => cache.originalFetched.length === 4);
// Favorite (10) first, then the window newest-first (1, 2, 3). File 4
// is outside the window and never pinned.
expect(cache.originalFetched).toEqual([10, 1, 2, 3]);
expect(pre.isPinned(10)).toBe(true);
expect(pre.isPinned(1)).toBe(true);
expect(pre.isPinned(4)).toBe(false);
});
it("drops a file from the pinned set when the window moves past it", () => {
const cols = [collection(1)];
const cache = new FakeCache();
const pre = new Precache({ originalsDays: 7 });
pre.bind(cache);
pre.update(records(cols, [file(1, 1, 0), file(2, 1, 3)]));
expect(pre.isPinned(2)).toBe(true);
// A newer file arrives; the window's newest end moves forward so the
// 3-day-old file 2 (now 13 days behind the newest) falls out.
pre.update(
records(cols, [file(3, 1, -10), file(1, 1, 0), file(2, 1, 3)]),
);
expect(pre.isPinned(3)).toBe(true);
expect(pre.isPinned(2)).toBe(false);
});
it("reports progress through status()", async () => {
const cols = [collection(1), collection(2, "favorites")];
const files = [file(1, 1, 0), file(2, 1, 1), file(10, 2, 0)];
const cache = new FakeCache();
const pre = new Precache({ originalsDays: 7 });
pre.bind(cache);
pre.update(records(cols, files));
const before = pre.status();
expect(before.thumbnailsTotal).toBe(3);
expect(before.thumbnailsCached).toBe(0);
expect(before.originalsPinned).toBe(3); // files 1, 2, 10 all in window
expect(before.originalsCached).toBe(0);
pre.start();
await until(
() =>
cache.thumbFetched.length === 3 &&
cache.originalFetched.length === 3,
);
const after = pre.status();
expect(after.thumbnailsCached).toBe(3);
expect(after.originalsCached).toBe(3);
});
it("honours the disable flags", async () => {
const cols = [collection(1)];
const files = [file(1, 1, 0)];
const cache = new FakeCache();
const pre = new Precache({ thumbnails: false, originals: false });
pre.bind(cache);
pre.update(records(cols, files));
pre.start();
await new Promise((r) => setTimeout(r, 20));
expect(cache.thumbFetched).toEqual([]);
expect(cache.originalFetched).toEqual([]);
expect(pre.isPinned(1)).toBe(false);
expect(pre.status().thumbnailsTotal).toBe(0);
});
});
// ---- Integration through the real ContentCache and Library ----
const enteFile = (id: number, collectionID: number): EnteFile =>
file(id, collectionID, 0);
describe("Precache eviction integration", () => {
let root: string;
beforeEach(() => {
root = mkdtempSync(join(tmpdir(), "quak-precache-evict-"));
});
afterEach(() => {
if (root && existsSync(root))
rmSync(root, { recursive: true, force: true });
});
it("never evicts a pinned original the precache put in place", async () => {
const cacheDir = join(root, "cache");
// File 1 is the favorites album's only file (pinned regardless of age);
// files 2 and 3 sit outside the latest-week window, so only file 1 is
// pinned. The by-id map serves the bytes for each fetch.
const cols = [collection(1), collection(2, "favorites")];
const files = [file(1, 2, 30), file(2, 1, 40), file(3, 1, 50)];
const byID = new Map<number, EnteFile>(files.map((f) => [f.id, f]));
const source: ContentSource = {
original: async ({ destination }) => {
await writeFile(destination, Buffer.alloc(10, 1));
return { bytesWritten: 10 };
},
thumbnail: async ({ destination }) => {
await writeFile(destination, Buffer.alloc(10, 1));
return { bytesWritten: 10 };
},
};
const statfs: StatFsFn = async () => ({
bsize: 1,
bavail: 1_000_000_000,
});
const pre = new Precache({ originalsDays: 7 });
const cache = new ContentCache({
pools: new RequestPools(),
source,
cacheDirectory: cacheDir,
getFile: (id) => byID.get(id),
statfs,
cacheOriginalsMaxBytes: 25, // holds two 10-byte originals
freeBelowBytes: 0,
isPinned: (id) => pre.isPinned(id),
});
pre.bind(cache);
pre.update(records(cols, files));
expect(pre.isPinned(1)).toBe(true);
expect(pre.isPinned(2)).toBe(false);
await cache.open();
// Fill three originals; the 25-byte cap forces an eviction on the
// third, and the pinned file 1 must survive it even though it is the
// least-recently-used.
await cache.original(1);
utimesSync(join(cacheDir, "originals", "1.jpg"), 1000, 1000); // oldest
await cache.original(2);
utimesSync(join(cacheDir, "originals", "2.jpg"), 2000, 2000);
await cache.original(3);
expect(existsSync(join(cacheDir, "originals", "1.jpg"))).toBe(true);
expect(existsSync(join(cacheDir, "originals", "2.jpg"))).toBe(false);
expect(existsSync(join(cacheDir, "originals", "3.jpg"))).toBe(true);
});
});
describe("Precache preemption", () => {
let root: string;
beforeEach(() => {
root = mkdtempSync(join(tmpdir(), "quak-precache-preempt-"));
});
afterEach(() => {
if (root && existsSync(root))
rmSync(root, { recursive: true, force: true });
});
it("lets an on-demand original preempt the background originals fill", async () => {
const byID = new Map<number, EnteFile>([
[1, enteFile(1, 1)],
[2, enteFile(2, 1)],
[3, enteFile(3, 1)],
]);
const finished: number[] = [];
let openGate!: () => void;
const gate = new Promise<void>((r) => (openGate = r));
let sawFirst!: () => void;
const firstStarted = new Promise<void>((r) => (sawFirst = r));
let started = 0;
const source: ContentSource = {
original: async ({ file: f, destination }) => {
if (++started === 1) sawFirst();
await gate;
await writeFile(destination, Buffer.alloc(10, 1));
finished.push(f.id);
return { bytesWritten: 10 };
},
thumbnail: async ({ destination }) => {
await writeFile(destination, Buffer.alloc(10, 1));
return { bytesWritten: 10 };
},
};
// One content slot, so file 1 holds it while 2 and 3 wait.
const cache = new ContentCache({
pools: new RequestPools({ contentConcurrency: 1 }),
source,
cacheDirectory: join(root, "cache"),
getFile: (id) => byID.get(id),
statfs: async () => ({ bsize: 1, bavail: 1_000_000_000 }),
freeBelowBytes: 0,
});
await cache.open();
const pA = cache.ensureOriginals({ fileIDs: [1] }); // background
await firstStarted; // file 1 now holds the only slot
const pB = cache.original(2); // on-demand, queued behind file 1
const pC = cache.ensureOriginals({ fileIDs: [3] }); // background, queued
await new Promise((r) => setTimeout(r, 5)); // let both enqueue
openGate();
await Promise.all([pA, pB, pC]);
// On-demand file 2 was served before the background file 3.
expect(finished).toEqual([1, 2, 3]);
});
});
describe("Precache through Library.open", () => {
let root: string;
beforeEach(() => {
root = mkdtempSync(join(tmpdir(), "quak-precache-lib-"));
});
afterEach(() => {
if (root && existsSync(root))
rmSync(root, { recursive: true, force: true });
});
class MockClient {
served = false;
whoami(): { email: string; userID: number } {
return { email: "u@example.com", userID: 7 };
}
async collectionsSince(): Promise<CollectionsPage> {
if (this.served) return { collections: [], deleted: [], cursor: 1 };
this.served = true;
return {
collections: [collection(1), collection(2, "favorites")],
deleted: [],
cursor: 1,
};
}
async filesSince(args: { collectionID: number }): Promise<FilesPage> {
const files =
args.collectionID === 1
? [enteFile(1, 1), enteFile(2, 1)]
: [enteFile(3, 2)];
return { files, deleted: [], cursor: 1 };
}
}
it("starts both precaches from open() and reports them in status()", async () => {
const source: ContentSource = {
original: async ({ destination }) => {
await writeFile(destination, Buffer.alloc(10, 1));
return { bytesWritten: 10 };
},
thumbnail: async ({ destination }) => {
await writeFile(destination, Buffer.alloc(10, 1));
return { bytesWritten: 10 };
},
};
// Each fill reports "done" once the cache has recorded its files. The
// source returning is not enough: the cache records a file only after
// it has checked it on disk.
const finished = new Set<string>();
let bothFinished!: () => void;
const precached = new Promise<void>((r) => (bothFinished = r));
const lib = await Library.open({
client: new MockClient(),
cacheDirectory: join(root, "cache"),
contentSource: source,
refreshIntervalSeconds: 3600,
onProgress: (e) => {
if (
e.status === "done" &&
(e.operation === "precacheThumbnails" ||
e.operation === "precacheOriginals")
) {
finished.add(e.operation);
if (finished.size === 2) bothFinished();
}
},
});
// Every file's thumbnail is precached; the favorite (file 3) and the
// week's files (1, 2) all have their originals precached.
await precached;
const status = lib.status();
expect(status.thumbnailsTotal).toBe(3);
expect(status.thumbnailsCached).toBe(3);
expect(status.originalsPinned).toBe(3);
expect(status.originalsCached).toBe(3);
await lib.close();
});
it.each(["thumbnail", "original"] as const)(
"close() resolves only after a running %s precache fetch has written its file",
async (kind) => {
// Only the fill under test runs, and each of its fetches waits
// until the test releases it.
let release!: () => void;
const held = new Promise<void>((r) => (release = r));
let fetchStarted!: (destination: string) => void;
const started = new Promise<string>((r) => (fetchStarted = r));
const fetch = async ({ destination }: { destination: string }) => {
fetchStarted(destination);
await held;
await writeFile(destination, Buffer.alloc(10, 1));
return { bytesWritten: 10 };
};
const unused = async () => {
throw new Error("this fill is turned off");
};
const source: ContentSource =
kind === "thumbnail"
? { original: unused, thumbnail: fetch }
: { original: fetch, thumbnail: unused };
const lib = await Library.open({
client: new MockClient(),
cacheDirectory: join(root, "cache"),
contentSource: source,
refreshIntervalSeconds: 3600,
precacheThumbnails: kind === "thumbnail",
precacheOriginals: kind === "original",
});
try {
const destination = await started;
let closed = false;
const closing = lib.close().then(() => {
closed = true;
});
await new Promise((r) => setTimeout(r, 20));
expect(closed).toBe(false);
release();
await closing;
expect(existsSync(destination)).toBe(true);
} finally {
release();
await lib.close();
}
},
);
});
+1 -1
View File
@@ -531,7 +531,7 @@ describe("Library exposes the read surface over its live store", () => {
expect(client.collectionsCalls).toBe(collectionsBefore);
expect(client.filesCalls).toBe(filesBefore);
} finally {
await lib.close();
lib.close();
}
});
});
+4 -4
View File
@@ -159,7 +159,7 @@ describe("Library.snapshot and Library.subscribe", () => {
for (const p of snap.photos) expect("key" in p).toBe(false);
for (const a of snap.albums) expect("key" in a).toBe(false);
} finally {
await lib.close();
lib.close();
}
});
@@ -219,7 +219,7 @@ describe("Library.snapshot and Library.subscribe", () => {
expect(change.refreshedAt).toBeGreaterThan(0);
} finally {
unsubscribe();
await lib.close();
lib.close();
}
});
@@ -251,7 +251,7 @@ describe("Library.snapshot and Library.subscribe", () => {
expect(changes).toEqual([]);
} finally {
unsubscribe();
await lib.close();
lib.close();
}
});
@@ -297,7 +297,7 @@ describe("Library.snapshot and Library.subscribe", () => {
);
expect(changes).toEqual([]);
} finally {
await lib.close();
lib.close();
}
});
});
+3 -73
View File
@@ -146,13 +146,10 @@ const buildSharedRawCollection = (
const buildRawFile = (
collectionKey: Uint8Array,
opts?: {
// Any JSON value; `undefined` leaves the title out of the metadata.
title?: unknown;
title?: string;
fileType?: number;
creationTime?: number;
info?: { fileSize?: number; thumbSize?: number };
// Replaces the whole metadata JSON value.
metadata?: unknown;
},
): RawEnteFile => {
const fileKey = sodium.crypto_secretbox_keygen();
@@ -161,8 +158,8 @@ const buildRawFile = (
collectionKey,
);
const defaultMetadata = {
title: opts && "title" in opts ? opts.title : "IMG_0001.jpg",
const metadata = {
title: opts?.title ?? "IMG_0001.jpg",
fileType: opts?.fileType ?? 0,
creationTime: opts?.creationTime ?? 1700000000000000,
modificationTime: 1700000000000000,
@@ -170,8 +167,6 @@ const buildRawFile = (
longitude: 2.3522,
hash: "abcdef1234567890",
};
const metadata =
opts && "metadata" in opts ? opts.metadata : defaultMetadata;
// File metadata is encrypted as a single-chunk secretstream blob
// (not secretbox). The decryptionHeader is the secretstream init header.
const metadataBytes = new TextEncoder().encode(JSON.stringify(metadata));
@@ -326,71 +321,6 @@ describe("model.decryptFile", () => {
expect(file.metadata.longitude).toBeCloseTo(2.3522);
});
it("reads a missing or non-string title as an empty string", () => {
// The server controls the metadata JSON. A title that is not a
// string must not reach code that builds file names from it.
const masterKey = sodium.crypto_secretbox_keygen();
const { collectionKey } = buildRawCollection(masterKey);
for (const title of [undefined, null, 42, ["a"], { x: "../y" }]) {
const file = decryptFile(
buildRawFile(collectionKey, { title }),
collectionKey,
);
expect(file.metadata.title).toBe("");
}
});
it("rejects metadata that is not a JSON object", () => {
const masterKey = sodium.crypto_secretbox_keygen();
const { collectionKey } = buildRawCollection(masterKey);
for (const metadata of [null, "IMG_0001.jpg", 7, []]) {
const raw = buildRawFile(collectionKey, { metadata });
expect(() => decryptFile(raw, collectionKey)).toThrow(
"file 200: metadata is not a JSON object",
);
}
});
it("reads the recorded content hash, joining an older live photo's two parts", () => {
const masterKey = sodium.crypto_secretbox_keygen();
const { collectionKey } = buildRawCollection(masterKey);
const hashOf = (metadata: Record<string, unknown>) =>
decryptFile(
buildRawFile(collectionKey, {
metadata: { title: "x", ...metadata },
}),
collectionKey,
).metadata.hash;
expect(hashOf({ fileType: 0, hash: "H" })).toBe("H");
expect(
hashOf({ fileType: 2, hash: "H", imageHash: "I", videoHash: "V" }),
).toBe("H");
expect(hashOf({ fileType: 2, imageHash: "I", videoHash: "V" })).toBe(
"I:V",
);
expect(hashOf({ fileType: 2, imageHash: "I" })).toBeUndefined();
expect(
hashOf({ fileType: 0, imageHash: "I", videoHash: "V" }),
).toBeUndefined();
expect(hashOf({ fileType: 0 })).toBeUndefined();
expect(hashOf({ fileType: 0, hash: 42 })).toBeUndefined();
expect(
hashOf({ fileType: 2, imageHash: "I", videoHash: 7 }),
).toBeUndefined();
// An empty string counts as absent, not as a hash to match.
expect(hashOf({ fileType: 0, hash: "" })).toBeUndefined();
expect(
hashOf({ fileType: 2, hash: "", imageHash: "I", videoHash: "V" }),
).toBe("I:V");
expect(
hashOf({ fileType: 2, imageHash: "", videoHash: "V" }),
).toBeUndefined();
expect(
hashOf({ fileType: 2, imageHash: "I", videoHash: "" }),
).toBeUndefined();
});
it("maps fileType numbers to FileType strings", () => {
// Ente uses: 0=image, 1=video, 2=livePhoto
const masterKey = sodium.crypto_secretbox_keygen();
+17 -12
View File
@@ -2,13 +2,13 @@
// failures are silent.
//
// Excluding too little: a worktree left under `.claude/` is copied into the
// image, vitest globs its `test/` tree as well as the real one, and the test
// phase runs the whole suite twice over while reporting success. A compiled
// `bin/quak` is ~100 MB of context nobody needs.
// image, vitest globs its `test/` tree as well as the real one, and the
// containerised `make check` runs the whole suite twice over while reporting
// success. A compiled `bin/quak` is ~100 MB of context nobody needs.
//
// Excluding too much: Prettier 3 reads `.gitignore` as a default ignore file,
// so dropping it from the context silently changes which files the lint
// phase's prettier check looks at compared to `make fmt-check` on the host.
// so dropping it from the context silently changes which files
// `make fmt-check` looks at inside the image compared to the host.
//
// Neither shows up as a build failure, so they are asserted here.
import { describe, expect, it } from "vitest";
@@ -47,13 +47,18 @@ describe(".dockerignore", () => {
expect(dockerignore).not.toContain(".gitignore");
});
// BuildKit lets a `Dockerfile.dockerignore` shadow the root one; such a
// file would silently give the build a different, unreviewed context —
// and eslint's flat config does not ignore dot-directories, so a stray
// `.claude/` worktree would be linted.
it("is not shadowed by a Dockerfile.dockerignore", () => {
expect(existsSync(join(repoRoot, "Dockerfile.dockerignore"))).toBe(
// Both images are built from this same context, and the lint image runs
// eslint and prettier across it. BuildKit lets a `<dockerfile>.dockerignore`
// shadow the root one for a single build; such a file would silently give
// the lint build a different, unreviewed context — and eslint's flat config
// does not ignore dot-directories, so a stray `.claude/` worktree would be
// linted.
it.each(["Dockerfile", "Dockerfile.lint"])(
"is not shadowed by a per-Dockerfile ignore file for %s",
(name) => {
expect(existsSync(join(repoRoot, `${name}.dockerignore`))).toBe(
false,
);
});
},
);
});
+4 -2
View File
@@ -1,7 +1,9 @@
// The package manifest promises three files that only exist after a build:
// `main`, `types`, and the `quak` binary. Nothing in the test suite used to
// look at them, and `make check` runs the test and lint phases but never the
// build, so `tsconfig.json` and `package.json` were free to drift apart. They
// look at them, and `make check` runs the suite and the lint container but
// never the build, so `tsconfig.json` and `package.json` were free to drift
// apart. (The formatting check is part of the lint container, not a step of
// its own; `test/packaging/lint-once.test.ts` is what holds that shape.) They
// did: `rootDir` was `./src` while `include` also pulled in `bin/**/*`, which
// is TS6059, and no build had succeeded for as long as that was true.
//
+184
View File
@@ -0,0 +1,184 @@
// Linting runs in Docker, one way, everywhere: `script/lint` builds
// `Dockerfile.lint`, which COPYs the repo into a digest-pinned image and runs
// eslint and prettier as build steps, so a successful build IS a clean lint.
//
// Three things can quietly undo that, and none of them shows up as a build
// failure, which is why they are asserted here:
//
// 1. Recursion. `script/check` calls `script/lint`, and `script/lint` is now a
// `docker build`. Anything that runs `make check` inside a container is
// therefore asking for Docker inside Docker, and CI breaks. The image built
// from `Dockerfile` runs the suite and the compile only; lint happens once,
// in `Dockerfile.lint`.
// 2. Cache. A lint build over an unchanged tree returns success in well under a
// second having linted nothing. The `LINT_EPOCH` guard is what forces the
// linter layers to execute, and it has to fail closed: an unset build
// argument is the empty string, which is a perfectly stable cache key, so an
// invocation that omits it must be rejected rather than served a cached
// green.
// 3. A host lint path surviving alongside the container one, which would let a
// lint result come from an unpinned local toolchain.
import { describe, expect, it } from "vitest";
import { readFileSync } from "node:fs";
import { fileURLToPath } from "node:url";
import { join } from "node:path";
const repoRoot = fileURLToPath(new URL("../../", import.meta.url));
const read = (name: string): string =>
readFileSync(join(repoRoot, name), "utf-8");
// The executable lines of a shell script or Dockerfile: comments carry the
// reasoning and frequently name the very commands these tests forbid, so they
// would otherwise trigger every assertion below.
const instructions = (name: string): string[] =>
read(name)
.split("\n")
.map((line) => line.trim())
.filter((line) => line !== "" && !line.startsWith("#"));
const lintScript = instructions("script/lint");
const dockerfileLint = instructions("Dockerfile.lint");
const dockerfile = instructions("Dockerfile");
const cibuild = instructions("script/cibuild");
const has = (lines: string[], pattern: RegExp): boolean =>
lines.some((line) => pattern.test(line));
describe("script/lint", () => {
it("lints by building Dockerfile.lint", () => {
expect(has(lintScript, /docker build .*-f Dockerfile\.lint/)).toBe(
true,
);
});
// The whole point of the ruling: no invocation of a linter against the
// working tree survives, so a lint verdict can only come from the pinned
// image.
it("runs no linter on the host", () => {
expect(has(lintScript, /eslint|prettier/)).toBe(false);
});
// Without a fresh epoch the build is served from cache in under a second,
// having linted nothing, and still exits 0.
it("passes a fresh LINT_EPOCH on every run", () => {
expect(
has(lintScript, /--build-arg LINT_EPOCH="\$\(date \+%s\)"/),
).toBe(true);
});
});
describe("Dockerfile.lint", () => {
// Tag references are server-mutable, so they are remote code execution.
it("pins its base image by digest", () => {
expect(has(dockerfileLint, /^FROM \S+@sha256:[0-9a-f]{64}/)).toBe(true);
});
it("runs eslint as a build step", () => {
expect(has(dockerfileLint, /^RUN .*eslint \./)).toBe(true);
});
it("runs prettier as a build step", () => {
expect(has(dockerfileLint, /^RUN .*prettier --check \./)).toBe(true);
});
// An unset ARG is the empty string, and an empty string is a perfectly
// stable cache key. Rejecting it is what stops a bare
// `docker build -f Dockerfile.lint .` from reporting a green it did not
// earn.
it("refuses to build without LINT_EPOCH", () => {
expect(has(dockerfileLint, /^ARG LINT_EPOCH$/)).toBe(true);
expect(
has(dockerfileLint, /^RUN \[ -n "\$LINT_EPOCH" \] \|\| exit 1$/),
).toBe(true);
});
// The guard only forces execution of the layers below it, so both linters
// have to sit after it. Layer order is the mechanism, not a style choice.
it("puts both linters below the epoch guard", () => {
const guard = dockerfileLint.findIndex((line) =>
/^RUN \[ -n "\$LINT_EPOCH" \]/.test(line),
);
const linters = dockerfileLint
.map((line, index) => ({ line, index }))
.filter(({ line }) => /^RUN .*(eslint|prettier)/.test(line));
expect(linters.length).toBeGreaterThan(0);
for (const { line, index } of linters) {
expect(
index,
`${line} must run below the LINT_EPOCH guard`,
).toBeGreaterThan(guard);
}
});
// Dependency installation is the slow layer and has nothing to do with the
// sources, so it caches separately: manifests first, sources afterwards.
it("copies the manifests before the sources", () => {
const manifests = dockerfileLint.findIndex((line) =>
/^COPY package\.json yarn\.lock/.test(line),
);
const sources = dockerfileLint.findIndex((line) =>
/^COPY \. \.$/.test(line),
);
expect(manifests).toBeGreaterThanOrEqual(0);
expect(sources).toBeGreaterThan(manifests);
});
// script/lint is a docker build; a lint step that shelled out to it would
// recurse.
it("does not call script/lint or make lint", () => {
expect(has(dockerfileLint, /make lint|script\/lint/)).toBe(false);
});
});
describe("Dockerfile", () => {
// `make check` runs script/lint, which is a docker build, so an image that
// ran it would need a Docker daemon inside the container.
it("does not run make check, make lint or script/lint", () => {
expect(
has(dockerfile, /make check|make lint|script\/(check|lint)/),
).toBe(false);
});
// The replaced lint stage took a `COPY --from=lint` dependency to order
// itself before the check stage. Dockerfile.lint is that stage now, and
// two definitions of how to lint is one too many.
it("has no lint stage", () => {
expect(has(dockerfile, /AS lint\b|--from=lint\b/)).toBe(false);
});
it("still runs the suite and the build under the epoch guard", () => {
expect(has(dockerfile, /^RUN make test$/)).toBe(true);
expect(has(dockerfile, /^RUN make build$/)).toBe(true);
expect(
has(dockerfile, /^RUN \[ -n "\$CHECK_EPOCH" \] \|\| exit 1$/),
).toBe(true);
});
});
describe("script/cibuild", () => {
// CI has to get both verdicts. Lint goes first so the fast failure is
// reported before the suite runs.
it("builds the lint image before the test and build image", () => {
const lint = cibuild.findIndex((line) => /\/lint"/.test(line));
const check = cibuild.findIndex((line) =>
/docker build .*CHECK_EPOCH/.test(line),
);
expect(lint).toBeGreaterThanOrEqual(0);
expect(check).toBeGreaterThan(lint);
});
});
describe("package.json", () => {
// `yarn lint` was a second, unpinned way to get a lint verdict, from
// whatever eslint the working tree happened to have installed.
it("exposes no host lint script", () => {
const pkg = JSON.parse(read("package.json")) as {
scripts: Record<string, string>;
};
expect(pkg.scripts.lint).toBeUndefined();
});
});
+503
View File
@@ -0,0 +1,503 @@
// `make check` used to run `prettier --check .` twice: once inside the lint
// container (`script/lint` builds `Dockerfile.lint`, which runs eslint and
// prettier as build steps) and once again on the host, because `script/check`
// also called `script/fmt-check`. Two passes, one verdict, and the host one is
// the weaker of the two — its prettier is whatever the working tree happens to
// have installed, while the container's is digest-pinned and installed under
// `--frozen-lockfile`.
//
// The fix was to delete the host call from `script/check` and `script/precommit`.
// Nothing about that fix is self-enforcing: anyone can wire `script/fmt-check`
// back in, or add a prettier step to a Dockerfile, and every build stays green
// while quietly doing the work twice again. So the count is asserted here
// rather than promised in a comment.
//
// The assertion is a static walk of the invocation graph, not a string match
// against one file. Starting from an entrypoint, it follows every edge the repo
// actually uses to reach another command — `run:` steps in the CI workflow,
// `"$SCRIPT_DIR/<name>"` and `script/<name>` into other scripts, `make <target>`
// through the Makefile shims, `yarn run <name>` through the `package.json`
// scripts, and `docker build -f <file>` into that Dockerfile's `RUN` steps — and
// counts the prettier invocations it finds. A prettier call added anywhere in
// that graph is therefore caught, wherever it is added.
//
// Two entrypoints are walked, because they cover different graphs: `make check`
// is what a developer runs, and `.gitea/workflows/check.yml` is what CI runs.
// The CI walk starts at the workflow file rather than at a hand-picked script,
// so "the path CI executes" is read out of the repo instead of assumed; it
// reaches `script/cibuild`, and through it the `Dockerfile` image that `make
// check` never touches. Walking only `make check` is how a duplicate prettier
// pass in `Dockerfile` stayed invisible.
//
// Undercounting is the failure mode that would make this test worthless. Three
// things guard against it: the walk is asserted to have reached the nodes that
// matter, an unresolvable or empty node is a thrown error rather than a quiet
// zero, and prettier is counted per occurrence rather than per line, so two
// invocations chained with `&&` cannot read as one.
import { describe, expect, it } from "vitest";
import { readFileSync } from "node:fs";
import { fileURLToPath } from "node:url";
import { join } from "node:path";
const repoRoot = fileURLToPath(new URL("../../", import.meta.url));
const read = (name: string): string =>
readFileSync(join(repoRoot, name), "utf-8");
// A backslash at end of line continues the command; the resolver has to see the
// whole invocation, since the interesting flags (`-f Dockerfile.lint`) can sit
// on the continuation.
const joinContinuations = (text: string): string[] => {
const joined: string[] = [];
for (const raw of text.split("\n")) {
const line = raw.trim();
const previous = joined[joined.length - 1];
if (previous !== undefined && previous.endsWith("\\")) {
joined[joined.length - 1] =
`${previous.slice(0, -1).trim()} ${line}`;
} else {
joined.push(line);
}
}
return joined;
};
// Comments are stripped everywhere. The headers of these scripts explain the
// duplication this test exists to prevent, and therefore name `prettier` and
// `script/fmt-check` repeatedly; counting them would make the test assert the
// prose instead of the behaviour.
const executable = (text: string): string[] =>
joinContinuations(text).filter(
(line) => line !== "" && !line.startsWith("#"),
);
// Every occurrence, not "does this line mention prettier": a line that reads
// `yarn run prettier --check . && yarn run prettier --check src` is two passes
// over the same tree, which is exactly the bug this file exists to catch, and
// counting it as one would hide it. `.prettierrc` and `.prettierignore` are not
// invocations and do not match, because `\b` requires a non-word character
// after the name.
const countPrettier = (line: string): number =>
(line.match(/\bprettier\b/g) ?? []).length;
// Makefile targets are thin shims (`check:` / tab / `@script/check`), so a
// `make <target>` edge has to resolve through them to keep "per `make check`"
// meaning what it says. Recipe lines are the tab-indented ones.
const makeRecipes = (): Map<string, string[]> => {
const recipes = new Map<string, string[]>();
let current: string | null = null;
for (const raw of read("Makefile").split("\n")) {
if (raw.startsWith("\t")) {
if (current !== null) {
recipes.get(current)?.push(raw.trim().replace(/^[@-]+/, ""));
}
continue;
}
const target = /^([a-z][a-z-]*)\s*:(?!=)/.exec(raw);
current = target === null ? null : target[1];
if (current !== null && !recipes.has(current)) {
recipes.set(current, []);
}
}
return recipes;
};
const recipes = makeRecipes();
const packageScripts = (): Record<string, string> => {
const pkg = JSON.parse(read("package.json")) as {
scripts?: Record<string, string>;
};
return pkg.scripts ?? {};
};
const scripts = packageScripts();
// Node keys: `script/<name>`, `docker:<Dockerfile>`, `make:<target>`,
// `yarn:<package.json script>`, `workflow:<CI workflow file>`.
const resolve = (node: string): string[] => {
if (node.startsWith("script/")) return executable(read(node));
if (node.startsWith("docker:")) {
return executable(read(node.slice("docker:".length)))
.filter((line) => line.startsWith("RUN "))
.map((line) => line.slice("RUN ".length));
}
// The `run:` steps of a workflow, in file order. `uses:` steps are actions,
// not commands, and have no edges into this repo's graph. A `run: |` block
// would resolve to the bare `|`, which reaches nothing and therefore fails
// the count rather than passing quietly.
if (node.startsWith("workflow:")) {
return executable(read(node.slice("workflow:".length)))
.filter((line) => /^-?\s*run:\s*\S/.test(line))
.map((line) => line.replace(/^-?\s*run:\s*/, ""));
}
if (node.startsWith("make:")) {
const target = node.slice("make:".length);
const recipe = recipes.get(target);
// A renamed or deleted target must be a loud failure: silently walking
// an empty recipe would report zero prettier invocations, which reads
// like the tidiest possible result.
if (recipe === undefined) {
throw new Error(`no such Makefile target: ${target}`);
}
return recipe;
}
if (node.startsWith("yarn:")) {
const name = node.slice("yarn:".length);
const script = scripts[name];
if (script === undefined) {
throw new Error(`no such package.json script: ${name}`);
}
return [script];
}
throw new Error(`unresolvable node: ${node}`);
};
// Same reasoning as the missing-target error, applied to every node kind: a
// node that resolves to no commands contributes zero prettier invocations and
// zero edges, which is indistinguishable from a clean result. Fail instead.
const commandsOf = (node: string): string[] => {
const commands = resolve(node);
if (commands.length === 0) {
throw new Error(`node resolved to no commands: ${node}`);
}
return commands;
};
const edgesOf = (line: string): string[] => {
const edges: string[] = [];
// `"$SCRIPT_DIR/lint"`, `"$ROOT/script/lint"` and a bare `script/lint` are
// all the same edge.
for (const match of line.matchAll(
/(?:\$SCRIPT_DIR|\$\{SCRIPT_DIR\}|script)\/([a-z][a-z-]*)/g,
)) {
edges.push(`script/${match[1]}`);
}
// Only real targets: `pkg_install gnumake make make make` in
// script/bootstrap is a package name, not an invocation of this Makefile.
for (const match of line.matchAll(/\bmake\s+([a-z][a-z-]*)/g)) {
if (recipes.has(match[1] ?? "")) edges.push(`make:${match[1]}`);
}
// Same rule for yarn: `yarn run prettier` is the linter itself (counted,
// not followed), `yarn run fmt-check` would be a package.json script that
// runs it indirectly.
for (const match of line.matchAll(/\byarn(?:\s+run)?\s+([a-z][a-z-]*)/g)) {
if ((match[1] ?? "") in scripts) edges.push(`yarn:${match[1]}`);
}
// The container lint pass lives behind a `docker build`; without following
// it the count would miss the one invocation that is supposed to survive.
if (/\bdocker\s+build\b/.test(line)) {
const file = /\s-f\s+(\S+)/.exec(line);
edges.push(`docker:${file === null ? "Dockerfile" : file[1]}`);
}
return edges;
};
interface Walk {
prettier: number;
reached: Set<string>;
}
// Repeated invocations must count repeatedly — running the same script twice is
// exactly the bug — so nodes are not deduplicated. The path stack is only there
// to turn a cycle into a loud failure instead of a hang.
//
// Counting and edge-following both happen for every line: a line that invokes
// prettier can also invoke something else, and skipping the edges of counted
// lines silently truncated the graph.
const walk = (node: string, path: string[] = [], into?: Walk): Walk => {
const result = into ?? { prettier: 0, reached: new Set<string>() };
if (path.includes(node)) {
throw new Error(`invocation cycle: ${[...path, node].join(" -> ")}`);
}
result.reached.add(node);
for (const line of commandsOf(node)) {
result.prettier += countPrettier(line);
for (const edge of edgesOf(line)) {
walk(edge, [...path, node], result);
}
}
return result;
};
describe("prettier runs exactly once per make check", () => {
const check = walk("make:check");
// The headline assertion, and the one the issue is about.
it("invokes prettier once for the whole of make check", () => {
expect(check.prettier).toBe(1);
});
// Guards against the count being 1 (or 0) because the walk never got
// anywhere. `make check` has to reach the suite, the lint script, and the
// Dockerfile whose build IS the lint verdict.
it.each(["script/check", "script/test", "script/lint", "Dockerfile.lint"])(
"reaches %s while counting",
(node) => {
const key = node.startsWith("script/") ? node : `docker:${node}`;
expect([...check.reached]).toContain(key);
},
);
// The one that survives is the container's, not the host's: that is the
// authoritative verdict, since a successful Dockerfile.lint build is what
// CI treats as proof of a clean tree.
it("keeps the surviving invocation inside the lint container", () => {
expect(walk("docker:Dockerfile.lint").prettier).toBe(1);
});
it("does not reach the host formatting check from make check", () => {
expect([...check.reached]).not.toContain("script/fmt-check");
});
});
describe("prettier runs exactly once per CI build", () => {
// Rooted at the workflow file, so this is the graph CI executes rather than
// the graph someone believed CI executes. `make check` cannot stand in for
// it: CI runs script/cibuild, which builds Dockerfile as well as
// Dockerfile.lint, and nothing under `make check` ever reads Dockerfile.
const ci = walk("workflow:.gitea/workflows/check.yml");
it("invokes prettier once for the whole CI build", () => {
expect(ci.prettier).toBe(1);
});
// script/cibuild is here because the workflow is asserted to run it;
// Dockerfile is here because it is the half of the CI graph that the
// `make check` walk cannot see.
it.each([
"script/cibuild",
"script/lint",
"docker:Dockerfile.lint",
"docker:Dockerfile",
])("reaches %s while counting", (node) => {
expect([...ci.reached]).toContain(node);
});
// The test and build image must not lint: linting is Dockerfile.lint's job,
// and a prettier step added here would be a second pass over the same tree
// for the same verdict — on the one path where it matters most.
it("keeps prettier out of the test and build image", () => {
expect(walk("docker:Dockerfile").prettier).toBe(0);
});
});
describe("the standalone entrypoints still do what their names say", () => {
// REPO_POLICIES.md requires both `make lint` and `make fmt-check` to exist
// and mean something. Dropping fmt-check from script/check must not turn it
// into a target nobody can use, and must not leave `make check` passing
// because both halves became no-ops.
it("still checks formatting under make fmt-check", () => {
expect(walk("make:fmt-check").prettier).toBe(1);
});
it("still checks formatting under make lint", () => {
expect(walk("make:lint").prettier).toBe(1);
});
});
describe("script/precommit", () => {
// Same duplication as script/check, same fix. The hook still catches a
// badly formatted tree before the commit lands, because script/lint is the
// container prettier run — that is the whole reason the host call could go.
it("checks formatting exactly once", () => {
expect(walk("script/precommit").prettier).toBe(1);
});
it("gets that check from the lint container", () => {
expect([...walk("script/precommit").reached]).toContain(
"docker:Dockerfile.lint",
);
});
});
// script/bootstrap installs the dependencies, and it has two install sites: one
// for the case where yarn has to be reached through nvm, and one for the case
// where yarn is already on PATH. A substring check against the whole file
// cannot tell them apart, so it reports the first and says nothing about the
// second — which is the one the containers take, because the pinned node image
// ships yarn. Both are resolved separately here.
const installBranches = (): { withoutYarn: string[]; withYarn: string[] } => {
const lines = executable(read("script/bootstrap"));
const open = lines.findIndex((line) =>
/^install_js_deps\s*\(\)/.test(line),
);
if (open === -1) {
throw new Error("script/bootstrap: no install_js_deps function");
}
const close = lines.indexOf("}", open);
const body = lines.slice(open + 1, close === -1 ? undefined : close);
const guard = body.findIndex((line) =>
/^if\b.*\bmissing yarn\b/.test(line),
);
const otherwise = body.indexOf("else", guard);
const end = body.indexOf("fi", otherwise);
if (guard === -1 || otherwise === -1 || end === -1) {
throw new Error(
"script/bootstrap: install_js_deps is not the expected " +
"if missing yarn / else / fi shape",
);
}
return {
withoutYarn: body.slice(guard + 1, otherwise),
withYarn: body.slice(otherwise + 1, end),
};
};
// Every `yarn install` in the given lines, with its flags, so an unpinned
// install cannot hide next to a pinned one.
const yarnInstalls = (lines: string[]): string[] =>
lines.flatMap((line) =>
[...line.matchAll(/\byarn install\b[^"'&|;]*/g)].map((match) =>
match[0].trim(),
),
);
describe("host and container prettier cannot disagree", () => {
// With the host pass gone from `make check`, `make fmt-check` is the only
// host-side formatting check left, and the container is the gate. The two
// must keep producing the same verdict on the same tree, or a developer
// running `make fmt-check` gets a green that CI then rejects.
//
// Three things make them agree, and all three are load-bearing:
it("pins the same prettier for both", () => {
const pkg = JSON.parse(read("package.json")) as {
devDependencies: Record<string, string>;
};
// An exact version, not a range: `^3.8.1` would let the container and
// the host resolve different builds with different formatting.
expect(pkg.devDependencies.prettier).toMatch(/^\d+\.\d+\.\d+$/);
});
it("installs from the lockfile on the branch the container takes", () => {
// Both images are FROM a node image, which ships yarn, so `missing
// yarn` is false and this is the branch that runs in the container.
const installs = yarnInstalls(installBranches().withYarn);
expect(installs).not.toHaveLength(0);
for (const install of installs) {
expect(install).toContain("--frozen-lockfile");
}
});
it("installs from the lockfile on the nvm branch too", () => {
// Not the container's branch, but it is the one a developer without
// yarn on PATH gets, and their prettier has to match the container's.
const installs = yarnInstalls(installBranches().withoutYarn);
expect(installs).not.toHaveLength(0);
for (const install of installs) {
expect(install).toContain("--frozen-lockfile");
}
});
it("runs script/bootstrap inside the lint container", () => {
// Without this the lockfile assertions above would be about a script
// the container never executes.
expect([...walk("docker:Dockerfile.lint").reached]).toContain(
"script/bootstrap",
);
});
it("keeps .gitignore in the build context", () => {
// Prettier 3 reads .gitignore as a default ignore file, so excluding it
// from the context would change which files the container checks.
const dockerignore = read(".dockerignore")
.split("\n")
.map((line) => line.trim());
expect(dockerignore).not.toContain(".gitignore");
});
});
describe("the walk cannot pass vacuously", () => {
// An earlier draft of this file computed a Makefile target as
// `node.slice("make:")` — a string where a number belongs, which coerces to
// NaN and made every target resolve to nothing. The count went to zero and
// an assertion of "not twice" would have been satisfied by a walk that had
// read nothing at all. Every way of reaching nothing is therefore an
// error here, and the ways are tested rather than assumed.
it("reports zero for a subgraph that does not run prettier", () => {
expect(walk("make:clean").prettier).toBe(0);
});
it("refuses a Makefile target that does not exist", () => {
expect(() => walk("make:no-such-target")).toThrow(
/no such Makefile target/,
);
});
it("refuses a package.json script that does not exist", () => {
expect(() => walk("yarn:no-such-script")).toThrow(
/no such package.json script/,
);
});
it("refuses a script that does not exist", () => {
expect(() => walk("script/no-such-script")).toThrow(/ENOENT/);
});
it("refuses a node that resolves to no commands", () => {
// .dockerignore has no RUN steps, standing in for a Dockerfile whose
// steps a restructure moved somewhere the resolver cannot see.
expect(() => walk("docker:.dockerignore")).toThrow(
/resolved to no commands/,
);
});
it("refuses a node kind it does not understand", () => {
expect(() => walk("nonsense")).toThrow(/unresolvable node/);
});
it("refuses to walk in circles", () => {
expect(() => walk("make:check", ["script/check"])).toThrow(
/invocation cycle/,
);
});
});
describe("the resolver reads what the shell would run", () => {
// Counting per line is how `yarn run prettier --check . && yarn run
// prettier --check src` read as a single invocation.
it("counts every prettier invocation on a line", () => {
expect(
countPrettier(
"yarn run prettier --check . && yarn run prettier --check src",
),
).toBe(2);
});
it("does not count the config files as invocations", () => {
expect(countPrettier("COPY .prettierrc .prettierignore ./")).toBe(0);
});
// The counting `continue` also dropped every edge that shared a line with a
// prettier call, so a whole subtree could be hidden behind one `&&`.
it("still follows the edges of a line that invokes prettier", () => {
expect(
edgesOf('yarn run prettier --check . && "$SCRIPT_DIR/lint"'),
).toContain("script/lint");
});
it("resolves every spelling of a script call to one node", () => {
expect(
edgesOf('"$SCRIPT_DIR/lint" "${SCRIPT_DIR}/test" script/fmt'),
).toEqual(["script/lint", "script/test", "script/fmt"]);
});
it("follows a bare docker build to Dockerfile and -f to its file", () => {
expect(edgesOf("docker build .")).toContain("docker:Dockerfile");
expect(edgesOf("docker build -f Dockerfile.lint .")).toContain(
"docker:Dockerfile.lint",
);
});
it("reads the run steps of the CI workflow and not its uses steps", () => {
expect(commandsOf("workflow:.gitea/workflows/check.yml")).toEqual([
"script/cibuild",
]);
});
});
-50
View File
@@ -1,50 +0,0 @@
// A checkout nested under `.claude/` must not add its tests to this suite.
// The test plants one in a temporary directory next to a real test file and
// asks vitest, with this repo's config, which test files it would run.
import { afterEach, describe, expect, it } from "vitest";
import { execFileSync } from "node:child_process";
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { fileURLToPath } from "node:url";
const repoRoot = fileURLToPath(new URL("../../", import.meta.url));
let root = "";
afterEach(() => {
rmSync(root, { recursive: true, force: true });
});
const writeTest = (path: string): void => {
mkdirSync(join(root, path, ".."), { recursive: true });
writeFileSync(
join(root, path),
'import { it } from "vitest";\nit("runs", () => {});\n',
);
};
describe("vitest.config.ts", () => {
it("does not collect tests from a checkout nested under .claude/", () => {
root = mkdtempSync(join(tmpdir(), "quak-nested-checkout-"));
writeTest("test/real.test.ts");
writeTest(".claude/worktrees/other/test/real.test.ts");
const output = execFileSync(
process.execPath,
[
join(repoRoot, "node_modules/vitest/vitest.mjs"),
"list",
"--filesOnly",
"--config",
join(repoRoot, "vitest.config.ts"),
"--root",
root,
],
{ cwd: root, encoding: "utf-8" },
);
const files = output.split("\n").filter((line) => line !== "");
expect(files).toEqual(["test/real.test.ts"]);
});
});
+4 -75
View File
@@ -149,11 +149,8 @@ describe("isRetryable: transport failures", () => {
});
it("retries an errno carried on the error itself", () => {
// Every errno the classifier names, so none can be reclassified
// unnoticed.
for (const code of [
"ECONNRESET",
"ECONNABORTED",
"ETIMEDOUT",
"EPIPE",
"ENOTFOUND",
@@ -161,8 +158,6 @@ describe("isRetryable: transport failures", () => {
"ECONNREFUSED",
"EHOSTUNREACH",
"ENETUNREACH",
"ENETRESET",
"ENETDOWN",
]) {
expect(isRetryable(errnoError(code))).toBe(true);
}
@@ -220,27 +215,6 @@ describe("isRetryable: transport failures", () => {
looped.cause = looped;
expect(isRetryable(looped)).toBe(false);
});
it("terminates on a cause chain that loops through two errors", () => {
const first: Error & { cause?: unknown } = new Error("first");
const second = new Error("second", { cause: first });
first.cause = second;
expect(isRetryable(first)).toBe(false);
});
it("reads the error and at most seven causes below it", () => {
// The walk is bounded at eight links. An errno at the eighth link is
// found; one at the ninth is not.
const buried = (causes: number): Error => {
let err = errnoError("ECONNRESET");
for (let i = 0; i < causes; i++) {
err = new Error(`wrapper ${i}`, { cause: err });
}
return err;
};
expect(isRetryable(buried(7))).toBe(true);
expect(isRetryable(buried(8))).toBe(false);
});
});
describe("isRetryable: stream truncation versus corruption", () => {
@@ -305,9 +279,10 @@ describe("isSafeToReplay", () => {
* that is not the whole question: the other half is "could the first
* attempt already have taken effect on the server?".
*
* The calls this guards are the `POST` and `PUT` requests listed in the
* README under "Endpoints used". A blind replay of some of them can do
* real damage, so they retry only on the failures that establish no TCP
* quak's non-idempotent calls are `/users/srp/create-session`,
* `/users/two-factor/verify` (which consumes one of a limited number of
* 2FA attempts) and `/files/thumbnail`. A blind replay of any of them can
* do real damage, so they retry only on the failures that establish no TCP
* connection to the server ever existed DNS produced no address, or the
* peer refused the connection and therefore that no request byte can
* have been transmitted.
@@ -358,52 +333,6 @@ describe("isSafeToReplay", () => {
).toBe(false);
expect(isSafeToReplay(new TypeError("fetch failed"))).toBe(false);
});
it("does not replay any other errno the classifier names", () => {
for (const code of [
"ECONNRESET",
"ECONNABORTED",
"ETIMEDOUT",
"EPIPE",
"EHOSTUNREACH",
"ENETUNREACH",
"ENETRESET",
"ENETDOWN",
]) {
expect(isSafeToReplay(errnoError(code))).toBe(false);
}
});
it("does not replay a chain that also shows the request may have gone out", () => {
// A connect errno somewhere in the chain is not enough: any other
// errno beside it is doubt, and doubt is not replayed.
const reset = Object.assign(
new Error("read ECONNRESET", { cause: errnoError("ECONNREFUSED") }),
{ code: "ECONNRESET" },
);
const mixed = new TypeError("fetch failed", { cause: reset });
expect(isRetryable(mixed)).toBe(true);
expect(isSafeToReplay(mixed)).toBe(false);
});
it("does not replay a chain longer than the walk reads", () => {
// Eight connect errnos, then a reset at the ninth link, below the
// limit. The walk never sees the reset, so it cannot rule it out.
const refusedChain = (below: Error | undefined): Error => {
let err = below;
for (let i = 0; i < 8; i++) {
err = Object.assign(new Error(`refused ${i}`, { cause: err }), {
code: "ECONNREFUSED",
});
}
return err as Error;
};
expect(isSafeToReplay(refusedChain(errnoError("ECONNRESET")))).toBe(
false,
);
// The same eight links with nothing below them are replayable.
expect(isSafeToReplay(refusedChain(undefined))).toBe(true);
});
});
// ---------------------------------------------------------------------------
+5 -39
View File
@@ -1,43 +1,9 @@
import { afterEach, describe, expect, it, vi } from "vitest";
import { readFileSync } from "node:fs";
import { describe, expect, it } from "vitest";
import { VERSION } from "../src/index.js";
const packageVersion = (
JSON.parse(
readFileSync(new URL("../package.json", import.meta.url), "utf-8"),
) as { version: string }
).version;
class ExitCalled extends Error {}
describe("version", () => {
const argv = process.argv;
afterEach(() => {
process.argv = argv;
vi.restoreAllMocks();
});
it("exports the version from package.json", () => {
expect(VERSION).toBe(packageVersion);
});
// Runs bin/quak.ts with --version. commander prints the version and then
// calls process.exit, which is stubbed to throw so the test survives.
it("reports the version from package.json in quak --version", async () => {
const printed: string[] = [];
vi.spyOn(process.stdout, "write").mockImplementation((chunk) => {
printed.push(String(chunk));
return true;
});
vi.spyOn(process, "exit").mockImplementation(() => {
throw new ExitCalled();
});
process.argv = ["node", "quak", "--version"];
await expect(import("../bin/quak.js")).rejects.toBeInstanceOf(
ExitCalled,
);
expect(printed.join("").trim()).toBe(packageVersion);
describe("quak", () => {
it("exports a version string", () => {
expect(typeof VERSION).toBe("string");
expect(VERSION.length).toBeGreaterThan(0);
});
});
+83 -318
View File
@@ -6,30 +6,17 @@
* working thumbnails, others return 404 or empty bodies. The tests
* verify that the detection and repair logic handles each case correctly.
*
* As of issue #52 both helpers take an open `Library` for enumeration and the
* `Client` for the API operations that stay unchanged (the thumbnail existence
* check, and the encrypt-and-upload path). `fixMissingThumbnails` reads each
* original through the library's content cache (`photo.original()`).
*
* `fixMissingThumbnails` is the most complex function in quak: it
* downloads the original file, generates a JPEG thumbnail with jpeg-js,
* encrypts it with secretstream push, gets a presigned upload URL,
* uploads to S3, and registers the new thumbnail with the API. The
* test verifies each step actually happened and the uploaded data is
* a valid encrypted blob that decrypts to a JPEG.
*
* It regenerates thumbnails for baseline JPEGs only, because `jpeg-js` decodes
* only JPEG. A non-JPEG image (PNG, HEIC) or a video is reported as "skipped
* (unsupported)" rather than crashing the decoder into an opaque failure
* (issue #17); the mixed test below locks that distinction down.
*/
import { existsSync, mkdtempSync, rmSync } from "node:fs";
import { join } from "node:path";
import { tmpdir } from "node:os";
import sodium from "libsodium-wrappers-sumo";
import * as jpegJs from "jpeg-js";
import { afterAll, beforeAll, describe, expect, it } from "vitest";
import { beforeAll, describe, expect, it } from "vitest";
import {
init,
toBase64,
@@ -40,7 +27,6 @@ import {
} from "../../src/crypto/index.js";
import { SRP, SrpServer } from "fast-srp-hap";
import { Client } from "../../src/client.js";
import { Library } from "../../src/library/index.js";
import {
listMissingThumbnails,
fixMissingThumbnails,
@@ -56,7 +42,6 @@ const TEST_EMAIL = "thumb@example.com";
const TEST_PASSWORD = "thumbpass";
const TEST_OPS = 2;
const TEST_MEM = 64 * 1024 * 1024;
const TEST_TIME = 1700000000000000;
interface ThumbMockState {
verifier: Buffer;
@@ -78,17 +63,8 @@ interface ThumbMockState {
}
let mock: ThumbMockState;
let tmpRoot: string;
// PNG signature bytes — enough for `fixMissingThumbnails` to recognise a
// non-JPEG image and skip it. It need not be a decodable PNG.
const PNG_BYTES = new Uint8Array([
0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a, 0x00, 0x00, 0x00, 0x0d,
]);
const buildThumbMock = async (opts?: {
extraFormats?: boolean;
}): Promise<ThumbMockState> => {
const buildThumbMock = async (): Promise<ThumbMockState> => {
const kekSalt = sodium.randombytes_buf(sodium.crypto_pwhash_SALTBYTES);
const kek = await deriveKEK(TEST_PASSWORD, kekSalt, TEST_OPS, TEST_MEM);
const loginSubKeyBytes = deriveLoginSubkey(kek);
@@ -126,6 +102,7 @@ const buildThumbMock = async (opts?: {
opsLimit: TEST_OPS,
};
// One collection with 3 files: ok thumbnail, empty thumbnail, 404 thumbnail
const collKey = sodium.crypto_secretbox_keygen();
const ckN = sodium.randombytes_buf(sodium.crypto_secretbox_NONCEBYTES);
const encCK = sodium.crypto_secretbox_easy(collKey, ckN, masterKey);
@@ -141,11 +118,10 @@ const buildThumbMock = async (opts?: {
encryptedName: toBase64(encCN),
nameDecryptionNonce: toBase64(cnN),
type: "album",
updationTime: TEST_TIME,
updationTime: 1700000000000000,
};
// Generate a real tiny JPEG via jpeg-js, used as the encrypted body of the
// JPEG files so a repair actually decodes and re-encodes real pixels.
// Generate a real tiny JPEG via jpeg-js
const w = 100;
const h = 80;
const pixels = new Uint8Array(w * h * 4);
@@ -155,32 +131,26 @@ const buildThumbMock = async (opts?: {
pixels[i + 2] = 0; // B
pixels[i + 3] = 255; // A
}
const tinyJpeg = new Uint8Array(
jpegJs.encode({ data: pixels, width: w, height: h }, 80).data,
);
const tinyJpeg = jpegJs.encode(
{ data: pixels, width: w, height: h },
80,
).data;
const fileKeys: Record<number, Uint8Array> = {};
const fileCiphertexts: Record<number, Uint8Array> = {};
const rawFiles: Record<string, unknown>[] = [];
// Build one raw file record: encrypt its metadata and its body under a
// fresh per-file key, and record the key and ciphertext for the mock to
// serve and for the test to verify against.
const makeRawFile = (
fileID: number,
fileType: number,
title: string,
body: Uint8Array,
): Record<string, unknown> => {
for (const fileID of [100, 101, 102]) {
const fk = sodium.crypto_secretstream_xchacha20poly1305_keygen();
fileKeys[fileID] = fk;
const fkN = sodium.randombytes_buf(sodium.crypto_secretbox_NONCEBYTES);
const encFK = sodium.crypto_secretbox_easy(fk, fkN, collKey);
const meta = JSON.stringify({
title,
fileType,
creationTime: TEST_TIME,
modificationTime: TEST_TIME,
title: `file-${fileID}.jpg`,
fileType: 0,
creationTime: 1700000000000000,
modificationTime: 1700000000000000,
});
const metaPush =
sodium.crypto_secretstream_xchacha20poly1305_init_push(fk);
@@ -191,17 +161,18 @@ const buildThumbMock = async (opts?: {
sodium.crypto_secretstream_xchacha20poly1305_TAG_FINAL,
);
// Encrypt the tiny JPEG as the file body
const filePush =
sodium.crypto_secretstream_xchacha20poly1305_init_push(fk);
const encFile = sodium.crypto_secretstream_xchacha20poly1305_push(
filePush.state,
body,
new Uint8Array(tinyJpeg),
null,
sodium.crypto_secretstream_xchacha20poly1305_TAG_FINAL,
);
fileCiphertexts[fileID] = encFile;
return {
rawFiles.push({
id: fileID,
collectionID: 1,
ownerID: 42,
@@ -215,33 +186,8 @@ const buildThumbMock = async (opts?: {
thumbnail: {
decryptionHeader: toBase64(sodium.randombytes_buf(24)),
},
// The encrypted size of the thumbnail the server records; large
// enough here that the default encoding fits.
info: { thumbSize: 1_000_000 },
updationTime: TEST_TIME,
};
};
// Three JPEG files: ok thumbnail, empty thumbnail, 404 thumbnail.
const rawFiles: Record<string, unknown>[] = [];
for (const fileID of [100, 101, 102]) {
rawFiles.push(makeRawFile(fileID, 0, `file-${fileID}.jpg`, tinyJpeg));
}
const thumbnailBehavior: Record<number, "ok" | "empty" | "404" | "500"> = {
100: "ok",
101: "empty",
102: "404",
};
// For the issue #17 mixed test: a non-JPEG image and a video, both with a
// missing (404) thumbnail so they surface in the missing list too.
if (opts?.extraFormats) {
rawFiles.push(makeRawFile(103, 0, "file-103.png", PNG_BYTES));
rawFiles.push(
makeRawFile(104, 1, "file-104.mp4", new Uint8Array([0, 0, 0, 1])),
);
thumbnailBehavior[103] = "404";
thumbnailBehavior[104] = "404";
updationTime: 1700000000000000,
});
}
return {
@@ -260,7 +206,11 @@ const buildThumbMock = async (opts?: {
filesByCollection: { 1: rawFiles },
fileCiphertexts,
fileKeys,
thumbnailBehavior,
thumbnailBehavior: {
100: "ok",
101: "empty",
102: "404",
},
uploadedThumbnails: [],
};
};
@@ -431,57 +381,6 @@ const countingFetch = (
return { fetch: fake as typeof globalThis.fetch, matched: () => matched };
};
// Open a library over a mock-backed client. As the CLI does for point commands,
// the background precache is off and the refresh interval is long, and the
// library client omits `fetchMLData` so no background ML fetch runs. The real
// `Client` is still used for the API operations the helpers perform directly.
const openLib = (client: Client): Promise<Library> =>
Library.open({
client: {
whoami: () => client.whoami(),
collectionsSince: (args) => client.collectionsSince(args),
filesSince: (args) => client.filesSince(args),
contentSource: () => client.contentSource(),
},
cacheDirectory: mkdtempSync(join(tmpRoot, "cache-")),
refreshIntervalSeconds: 3600,
precacheThumbnails: false,
precacheOriginals: false,
});
/** The mock's raw record for one file, for a test to change before login. */
const rawFile = (m: ThumbMockState, fileID: number): Record<string, unknown> =>
m.filesByCollection[1]!.find((f) => f.id === fileID)!;
/** Replace the original the mock serves for one file. */
const replaceOriginal = (
m: ThumbMockState,
fileID: number,
body: Uint8Array,
): void => {
const push = sodium.crypto_secretstream_xchacha20poly1305_init_push(
m.fileKeys[fileID]!,
);
m.fileCiphertexts[fileID] =
sodium.crypto_secretstream_xchacha20poly1305_push(
push.state,
body,
null,
sodium.crypto_secretstream_xchacha20poly1305_TAG_FINAL,
);
rawFile(m, fileID).file = { decryptionHeader: toBase64(push.header) };
};
const isOriginalDownload = (url: string): boolean =>
url.includes("files.ente.io") || url.includes("/files/download/");
const login = (fetch: typeof globalThis.fetch, retry?: RetryOptions) =>
Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: retry ? { fetch, retry } : { fetch },
});
// ---------------------------------------------------------------------------
// Tests
// ---------------------------------------------------------------------------
@@ -490,21 +389,17 @@ beforeAll(async () => {
await init();
await sodium.ready;
mock = await buildThumbMock();
tmpRoot = mkdtempSync(join(tmpdir(), "quak-thumb-test-"));
});
afterAll(() => {
if (tmpRoot && existsSync(tmpRoot))
rmSync(tmpRoot, { recursive: true, force: true });
});
describe("listMissingThumbnails", () => {
it("identifies files with empty and 404 thumbnails, ignores working ones", async () => {
const client = await login(buildThumbFetch(mock));
const lib = await openLib(client);
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildThumbFetch(mock) },
});
const missing = await listMissingThumbnails(lib, client);
lib.close();
const missing = await listMissingThumbnails(client);
// File 100 has a working thumbnail → not reported
// File 101 has an empty thumbnail → reported
@@ -541,11 +436,13 @@ describe("listMissingThumbnails", () => {
buildThumbFetch(failingMock),
(url) => url.includes("thumbnails.ente.io") && url.includes("102"),
);
const client = await login(counted.fetch, { ...noWait });
const lib = await openLib(client);
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: counted.fetch, retry: { ...noWait } },
});
const missing = await listMissingThumbnails(lib, client);
lib.close();
const missing = await listMissingThumbnails(client);
// Only the genuinely empty thumbnail is reported.
expect(missing.map((m) => m.fileID)).toEqual([101]);
@@ -578,11 +475,13 @@ describe("listMissingThumbnails", () => {
return inner(input, init);
}) as typeof globalThis.fetch;
const client = await login(fetch, { ...noWait });
const lib = await openLib(client);
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch, retry: { ...noWait } },
});
const missing = await listMissingThumbnails(lib, client);
lib.close();
const missing = await listMissingThumbnails(client);
expect(missing.map((m) => m.fileID)).toEqual([101]);
expect(thumbRequests).toBe(4);
@@ -601,55 +500,32 @@ describe("listMissingThumbnails", () => {
mockWithDupes.filesByCollection[2] =
mockWithDupes.filesByCollection[1]!;
const client = await login(buildThumbFetch(mockWithDupes));
const lib = await openLib(client);
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildThumbFetch(mockWithDupes) },
});
const missing = await listMissingThumbnails(lib, client);
lib.close();
const missing = await listMissingThumbnails(client);
// Should still be 2, not 4 (each file checked only once)
expect(missing.length).toBe(2);
});
it("skips a file another account owns without fetching its thumbnail", async () => {
const otherMock = await buildThumbMock();
rawFile(otherMock, 102).ownerID = 7;
const logs: string[] = [];
const counted = countingFetch(
buildThumbFetch(otherMock),
(url) => url.includes("thumbnails.ente.io") && url.includes("102"),
);
const client = await login(counted.fetch);
const lib = await openLib(client);
const missing = await listMissingThumbnails(lib, client, (msg) =>
logs.push(msg),
);
lib.close();
expect(missing.map((m) => m.fileID)).toEqual([101]);
expect(counted.matched()).toBe(0);
expect(
logs.some(
(l) =>
l.includes("Skipping file-102.jpg") &&
l.includes("another account"),
),
).toBe(true);
});
});
describe("fixMissingThumbnails", () => {
it("downloads original, generates thumbnail, encrypts, uploads, and registers", async () => {
const fixMock = await buildThumbMock();
const client = await login(buildThumbFetch(fixMock));
const lib = await openLib(client);
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildThumbFetch(fixMock) },
});
const results = await fixMissingThumbnails(lib, client, [101]);
lib.close();
const results = await fixMissingThumbnails(client, [101]);
expect(results.length).toBe(1);
expect(results[0]!.status).toBe("fixed");
expect(results[0]!.success).toBe(true);
expect(results[0]!.fileID).toBe(101);
expect(results[0]!.title).toBe("file-101.jpg");
expect(results[0]!.collection).toBe("Photos");
@@ -679,163 +555,48 @@ describe("fixMissingThumbnails", () => {
it("reports failure for nonexistent file IDs without crashing", async () => {
const fixMock = await buildThumbMock();
const client = await login(buildThumbFetch(fixMock));
const lib = await openLib(client);
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildThumbFetch(fixMock) },
});
const results = await fixMissingThumbnails(lib, client, [999]);
lib.close();
const results = await fixMissingThumbnails(client, [999]);
expect(results.length).toBe(1);
expect(results[0]!.status).toBe("failed");
expect(results[0]!.success).toBe(false);
expect(results[0]!.fileID).toBe(999);
expect(results[0]!.reason).toContain("not found");
expect(results[0]!.error).toContain("not found");
});
it("continues after one file fails and reports mixed results", async () => {
const fixMock = await buildThumbMock();
// Make file 102 fail by removing its ciphertext so the download 404s.
// Make file 102 fail by removing its ciphertext so download fails
delete fixMock.fileCiphertexts[102];
const client = await login(buildThumbFetch(fixMock));
const lib = await openLib(client);
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildThumbFetch(fixMock) },
});
const results = await fixMissingThumbnails(lib, client, [101, 102]);
lib.close();
const results = await fixMissingThumbnails(client, [101, 102]);
expect(results.length).toBe(2);
const success = results.find((r) => r.fileID === 101)!;
const failure = results.find((r) => r.fileID === 102)!;
expect(success.status).toBe("fixed");
expect(failure.status).toBe("failed");
});
it("skips a non-JPEG image and a video as unsupported, not failed (issue #17)", async () => {
// A PNG and a video both throw inside the JPEG decoder. The helper must
// recognise them up front and report "skipped", distinct from a genuine
// "failed", and must not upload anything for them. The JPEG in the same
// batch is still repaired.
const fixMock = await buildThumbMock({ extraFormats: true });
const client = await login(buildThumbFetch(fixMock));
const lib = await openLib(client);
const results = await fixMissingThumbnails(
lib,
client,
[101, 103, 104],
);
lib.close();
const jpeg = results.find((r) => r.fileID === 101)!;
const png = results.find((r) => r.fileID === 103)!;
const video = results.find((r) => r.fileID === 104)!;
expect(jpeg.status).toBe("fixed");
// The PNG is a still image but not a JPEG: skipped only after its bytes
// are inspected.
expect(png.status).toBe("skipped");
expect(png.reason).toContain("JPEG");
// The video is skipped from its type alone, before any download.
expect(video.status).toBe("skipped");
expect(video.reason).toContain("video");
// Only the JPEG was uploaded; the two skipped files touched no upload.
expect(fixMock.uploadedThumbnails.length).toBe(1);
expect(fixMock.uploadedThumbnails[0]!.fileID).toBe(101);
});
it("skips a file another account owns without downloading it", async () => {
// The server accepts a thumbnail only from the file's owner.
const fixMock = await buildThumbMock();
rawFile(fixMock, 101).ownerID = 7;
const counted = countingFetch(
buildThumbFetch(fixMock),
isOriginalDownload,
);
const client = await login(counted.fetch);
const lib = await openLib(client);
const results = await fixMissingThumbnails(lib, client, [101]);
lib.close();
expect(results[0]!.status).toBe("skipped");
expect(results[0]!.reason).toContain("another account");
expect(counted.matched()).toBe(0);
expect(fixMock.uploadedThumbnails.length).toBe(0);
});
it("skips a file whose recorded thumbnail size is 0 without downloading it", async () => {
// The server refuses a thumbnail larger than the one it records, and
// no thumbnail is 0 bytes.
const fixMock = await buildThumbMock();
rawFile(fixMock, 101).info = { thumbSize: 0 };
const counted = countingFetch(
buildThumbFetch(fixMock),
isOriginalDownload,
);
const client = await login(counted.fetch);
const lib = await openLib(client);
const results = await fixMissingThumbnails(lib, client, [101]);
lib.close();
expect(results[0]!.status).toBe("skipped");
expect(results[0]!.reason).toContain("recorded thumbnail size is 0");
expect(counted.matched()).toBe(0);
expect(fixMock.uploadedThumbnails.length).toBe(0);
});
it("re-encodes smaller until the thumbnail fits the recorded size", async () => {
// A noisy 400x300 JPEG, which the default encoding (quality 50, not
// resized because it is under 720 px) cannot compress below the size
// recorded here: one byte less than that encoding's ciphertext.
const fixMock = await buildThumbMock();
const w = 400;
const h = 300;
const noisy = new Uint8Array(
jpegJs.encode(
{
data: sodium.randombytes_buf(w * h * 4),
width: w,
height: h,
},
90,
).data,
);
replaceOriginal(fixMock, 101, noisy);
const decoded = jpegJs.decode(noisy, {
useTArray: true,
formatAsRGBA: true,
});
const defaultSize =
jpegJs.encode(decoded, 50).data.length +
sodium.crypto_secretstream_xchacha20poly1305_ABYTES;
const recordedSize = defaultSize - 1;
rawFile(fixMock, 101).info = { thumbSize: recordedSize };
const client = await login(buildThumbFetch(fixMock));
const lib = await openLib(client);
const results = await fixMissingThumbnails(lib, client, [101]);
lib.close();
expect(results[0]!.status).toBe("fixed");
const upload = fixMock.uploadedThumbnails[0]!;
expect(upload.ciphertext.length).toBeLessThanOrEqual(recordedSize);
const decrypted = decryptBlob(
upload.ciphertext,
fromBase64(upload.decryptionHeader),
fixMock.fileKeys[101]!,
);
expect(decrypted[0]).toBe(0xff);
expect(decrypted[1]).toBe(0xd8);
expect(success.success).toBe(true);
expect(failure.success).toBe(false);
});
});
describe("Client.getApiClient", () => {
it("returns the ApiClient when logged in", async () => {
const client = await login(buildThumbFetch(mock));
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildThumbFetch(mock) },
});
const api = client.getApiClient();
expect(api).toBeDefined();
@@ -843,7 +604,11 @@ describe("Client.getApiClient", () => {
});
it("throws after logout", async () => {
const client = await login(buildThumbFetch(mock));
const client = await Client.login({
email: TEST_EMAIL,
password: TEST_PASSWORD,
apiOptions: { fetch: buildThumbFetch(mock) },
});
client.logout();
expect(() => client.getApiClient()).toThrow(/logged out/);
-10
View File
@@ -1,10 +0,0 @@
import { configDefaults, defineConfig } from "vitest/config";
// vitest does not read .gitignore when looking for tests. A checkout nested
// under .claude/ has its own test/ tree, and without this exclude the suite
// runs once per nested checkout and still reports success.
export default defineConfig({
test: {
exclude: [...configDefaults.exclude, ".claude/**"],
},
});
+8 -6
View File
@@ -528,6 +528,13 @@
resolved "https://registry.yarnpkg.com/@types/json-schema/-/json-schema-7.0.15.tgz#596a1747233694d50f6ad8a7869fcb6f56cf5841"
integrity sha512-5+fP8P8MFNC+AyZCDxrB2pkZFPGzqQWUzpSeuuVLvm8VMcorNYavBqoFcxK8bQz4Qsbn4oUEEem4wDLfcysGHA==
"@types/libsodium-wrappers-sumo@0.8.2":
version "0.8.2"
resolved "https://registry.yarnpkg.com/@types/libsodium-wrappers-sumo/-/libsodium-wrappers-sumo-0.8.2.tgz#488e8747fbb982fe901020b5afeaddfa63da6830"
integrity sha512-uFOBpg/r21hExVlh2ty8YpDfSR+Yy3Jn8XS4+SSjitbhTxdYq+pBz/49XRxyUFe8SzqujHf/Wu0/O4d+FUtNfQ==
dependencies:
libsodium-wrappers-sumo "*"
"@types/node@22.18.13":
version "22.18.13"
resolved "https://registry.yarnpkg.com/@types/node/-/node-22.18.13.tgz#a037c4f474b860be660e05dbe92a9ef945472e28"
@@ -1069,11 +1076,6 @@ fastq@^1.6.0:
dependencies:
reusify "^1.0.4"
fflate@0.8.3:
version "0.8.3"
resolved "https://registry.yarnpkg.com/fflate/-/fflate-0.8.3.tgz#bc27d8eb30343d4d512abb03480202ce65d825fc"
integrity sha512-tbZNuJrLwGUp3zshBtdy4W+ORxZuIh8a5ilyIEQDC5rY1f3U20JMry0Ll3WBzU58EZKsEuJFXhb5gwv8CsPvgA==
file-entry-cache@^8.0.0:
version "8.0.0"
resolved "https://registry.yarnpkg.com/file-entry-cache/-/file-entry-cache-8.0.0.tgz#7787bddcf1131bffb92636c69457bbc0edd6d81f"
@@ -1247,7 +1249,7 @@ libsodium-sumo@^0.8.0:
resolved "https://registry.yarnpkg.com/libsodium-sumo/-/libsodium-sumo-0.8.4.tgz#6d4687781fa0ad398af14a7df872d5c27cf8cd31"
integrity sha512-TMtHShQfVVsaxDygyapvUC3o7YsPgXa/hRWeIgzyFz6w5k/1hirGptCxp1U7XwW3rCskaTTYKgV10v86UiGgNw==
libsodium-wrappers-sumo@0.8.4:
libsodium-wrappers-sumo@*, libsodium-wrappers-sumo@0.8.4:
version "0.8.4"
resolved "https://registry.yarnpkg.com/libsodium-wrappers-sumo/-/libsodium-wrappers-sumo-0.8.4.tgz#6656a3e7e0551ecce08ddee4bfb501a092eac6fa"
integrity sha512-ql7hcgulKZ3ekfa2DGAogcCKsWU0diA/0nArz1CFzh93WQdb46/Kj18ka/Hifq6uA3Ush34Pc6vU/6HXeRwUkg==