diff --git a/.dockerignore b/.dockerignore index 4beaf77..e2a5e34 100644 --- a/.dockerignore +++ b/.dockerignore @@ -1,6 +1,6 @@ # Mirrors .gitignore, with one deliberate exception: .gitignore itself stays # in the build context, because prettier 3 reads it as a default ignore file -# and dropping it would change what `make fmt-check` sees inside the image. +# and dropping it would change what the lint phase's prettier check sees. # VCS .git diff --git a/Dockerfile b/Dockerfile index a194ffd..144308a 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,28 +1,69 @@ -# Test and build image: the suite, then the compile. +# Lint phase. The linters are invoked directly rather than through `make +# lint` or `script/lint`, which are themselves a docker build and would +# recurse into a daemon that does not exist in a build step. # -# Linting deliberately does not happen here. `script/lint` is a build of -# Dockerfile.lint, and `script/check` calls `script/lint`, so running -# `make check` in this image would mean running `docker build` inside a -# container. Lint runs exactly once, in Dockerfile.lint; script/cibuild -# builds that first and this second. # node 22.22.0 on Alpine 3.23.3 (node:22-alpine), 2026-08-09 -FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS check +FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS lint + WORKDIR /app COPY script/ script/ COPY package.json yarn.lock ./ RUN script/bootstrap + COPY . . -# CHECK_EPOCH is a cache buster: without it Docker serves the test layer from -# cache on an unchanged tree, the suite never executes, and the build still -# exits 0. The guard makes an absent argument a hard failure — an unset ARG -# is the empty string, which is a perfectly stable cache key, so a plain -# `docker build .` would otherwise still get the false green. Fail closed. -ARG CHECK_EPOCH -RUN [ -n "$CHECK_EPOCH" ] || exit 1 -RUN make test +RUN yarn run eslint . +RUN yarn run prettier --check . + +# Test phase, same shape and for the same reason. The suite runs without +# verbose output first and is rerun verbosely only if it fails; the timeout +# catches a hung test. +# +# node 22.22.0 on Alpine 3.23.3 (node:22-alpine), 2026-08-09 +FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS test + +WORKDIR /app + +COPY script/ script/ +COPY package.json yarn.lock ./ +RUN script/bootstrap + +COPY . . + +# Unlike the template, the suite runs as the image's non-root `node` user: +# root ignores directory permissions, so the tests of a destination that is +# not writable would otherwise fail. vitest writes into /app. +RUN chown -R node:node /app +USER node + +RUN timeout 90 yarn run vitest run --reporter=dot || \ + { echo "--- Rerunning with verbose for details ---"; \ + timeout 90 yarn run vitest run --reporter=verbose; exit 1; } + +# Build stage, and the last stage: a plain `docker build .` names no target +# and so builds this one. Nothing is wanted from the two phases above; the +# copies are what make BuildKit build them first, so this image cannot be +# produced unless lint and test passed. A stage appended after this one +# would drop all three out of a plain build. +# +# node 22.22.0 on Alpine 3.23.3 (node:22-alpine), 2026-08-09 +FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 + +WORKDIR /app + +COPY --from=lint /app/package.json /dev/null +COPY --from=test /app/package.json /dev/null + +COPY script/ script/ +COPY package.json yarn.lock ./ +RUN script/bootstrap + +COPY . . + +# The version is computed on the host and passed in, because +# .dockerignore excludes .git. +ARG VERSION=dev +LABEL org.opencontainers.image.version="${VERSION}" -ARG CHECK_EPOCH -RUN [ -n "$CHECK_EPOCH" ] || exit 1 RUN make build diff --git a/Dockerfile.lint b/Dockerfile.lint deleted file mode 100644 index 0ee2a9a..0000000 --- a/Dockerfile.lint +++ /dev/null @@ -1,35 +0,0 @@ -# Lint image: every lint run happens here, and nowhere else. The repo is -# COPYed into a digest-pinned image and the linters run as build steps, so a -# successful build IS a clean lint. `script/lint` does nothing but build this -# file, which also works where the docker daemon is remote and bind mounts are -# impossible. Nothing that runs inside a container may call `script/lint`: -# that is why Dockerfile no longer runs `make check`. -# node 22.22.0 on Alpine 3.23.3 (node:22-alpine), 2026-08-09 -FROM node@sha256:e4bf2a82ad0a4037d28035ae71529873c069b13eb0455466ae0bc13363826e34 AS lint -WORKDIR /app - -# Manifests before sources, so the dependency install layer stays cached -# until package.json or yarn.lock changes. script/bootstrap ends in -# `yarn install --frozen-lockfile`; the lint steps below are deliberately -# not cached. -COPY script/ script/ -COPY package.json yarn.lock ./ -RUN script/bootstrap - -COPY . . - -# LINT_EPOCH is a cache buster, with the same fail-closed contract as -# CHECK_EPOCH in Dockerfile. No lint cache is wanted: on an unchanged tree -# Docker serves the linter layers in well under a second, having linted -# nothing, and the build still exits 0. The guard makes an absent argument a -# hard failure — an unset ARG is the empty string, which is a perfectly -# stable cache key, so a plain `docker build -f Dockerfile.lint .` would -# otherwise get exactly that false green. Every layer below this one is a -# child of the guard, so a fresh epoch forces all of them to execute. -ARG LINT_EPOCH -RUN [ -n "$LINT_EPOCH" ] || exit 1 - -# The linters are invoked directly rather than through `make lint`, because -# `make lint` is the build of this file. -RUN yarn run eslint . -RUN yarn run prettier --check . diff --git a/README.md b/README.md index 38b8039..3903483 100644 --- a/README.md +++ b/README.md @@ -73,7 +73,7 @@ if (photo) { console.log(`original at ${path}`); } -lib.close(); +await lib.close(); ``` The lower-level `Client` (login, session serialization, and the raw @@ -97,20 +97,18 @@ alpine. We provide: - `script/build` — compile the TypeScript sources into `dist/`, then verify that the entrypoints `package.json` declares (`main`, `types`, `bin`) are among the files the compiler wrote, and make the CLI executable (our own extension) -- `script/test` — run the test suite (vitest, hard-capped at 30s where `timeout` - is available, verbose rerun on failure) -- `script/lint` — run eslint and a prettier check, by building - `Dockerfile.lint`; requires docker (see Linting below) +- `script/test` — run the test suite, by building the `test` phase of the + `Dockerfile` (vitest, 90s timeout, verbose rerun on failure); requires docker +- `script/lint` — run eslint and a prettier check, by building the `lint` phase + of the `Dockerfile`; requires docker (see Linting and testing below) - `script/fmt` — format all files with prettier (writes) - `script/fmt-check` — check formatting on the host (read-only); standalone, and not called by `script/check` or `script/precommit`, because `script/lint` - already checks formatting in the container (see Linting below) + already checks formatting in the container - `script/check` — run all checks: `test`, `lint` (our own extension) -- `script/docker` — build the test and build image, tagged via - `script/projectname` -- `script/cibuild` — cd to the repo root and build both images (what CI runs): - `script/lint` first, then the `Dockerfile` image, which runs `make test` and - `make build` +- `script/docker` — build the image, tagged via `script/projectname` +- `script/cibuild` — build the image (what CI runs); its last stage depends on + the `lint` and `test` phases, so this one build lints, tests and compiles - `script/precommit` — run by the git pre-commit hook (our own extension); runs `script/lint`, which checks both lint and formatting, but deliberately not the tests, so the TDD red-phase commit can land @@ -119,50 +117,32 @@ alpine. We provide: `make hooks` installs the pre-commit hook that runs `script/precommit`. -### Linting +### Linting and testing -Linting runs in a container, one way, everywhere. `script/lint` builds -`Dockerfile.lint`, which copies the repo into a digest-pinned node image and -runs eslint and prettier as build steps, so a successful build is a clean lint. -There is no host lint path: docker is required to lint, and that also works -where the docker daemon is remote and bind mounts are impossible. +Linting and testing are phases of the `Dockerfile`. The `lint` phase copies the +repo into a digest-pinned node image and runs eslint and `prettier --check .`; +the `test` phase does the same with the suite. `script/lint` and `script/test` +each build one phase with `docker build --no-cache --target `. There is +no host lint or test path: docker is required, and that also works where the +docker daemon is remote and bind mounts are impossible. -The formatting check is part of that, not a step beside it. `script/check` and -`script/precommit` therefore call `script/lint` and stop; neither calls -`script/fmt-check` as well, which would run prettier a second time over the same -tree for the same verdict — and the weaker of the two, since the host's prettier -is whatever the working tree has installed. So `make check` and the pre-commit -hook both still fail on a badly formatted tree, and prettier runs exactly once -in each. `test/packaging/lint-once.test.ts` asserts that count by walking the -invocation graph, so a second pass cannot creep back in unnoticed. +The last stage of the `Dockerfile` compiles the package, and it copies a file +from each phase, so it cannot be built unless lint and the tests pass. That is +why `script/cibuild` is a single `docker build`: it runs lint and the tests once +each and then compiles. +Every `docker build` in `script/` passes `--no-cache`. On an unchanged tree +Docker would otherwise serve the lint and test steps from cache, nothing would +run, and the build would still exit 0. + +The formatting check is part of the `lint` phase, not a step beside it, so +`script/check` and `script/precommit` do not call `script/fmt-check` as well; +that would run prettier a second time over the same tree for the same verdict. `script/fmt-check` remains as a standalone entrypoint for asking the formatting -question on its own, without docker and without the rest of lint. Its verdict -cannot drift from the container's: prettier is pinned to an exact version, -installed from `yarn.lock` under `--frozen-lockfile` in both places, and reads -`.gitignore` as its default ignore file — which is why `.dockerignore` -deliberately keeps `.gitignore` in the build context. - -Lint happens in exactly one place, which constrains the rest of the build. -`script/check` calls `script/lint`, so `make check` cannot run inside a -container without asking for docker inside docker. The image built from -`Dockerfile` therefore runs `make test` and `make build` and does not lint; -`script/cibuild` builds `Dockerfile.lint` first and that image second, so CI -gets both verdicts. - -### Build epochs - -`script/lint` passes `--build-arg LINT_EPOCH="$(date +%s)"`, and `script/docker` -and `script/cibuild` pass `--build-arg CHECK_EPOCH="$(date +%s)"`. Both -Dockerfiles refuse to build without their argument. This is deliberate: on an -unchanged tree Docker would otherwise serve the linter and test layers from -cache, so nothing would run and the build would still exit 0 — a lint build over -an untouched tree returns success in well under a second, having linted nothing. -A changing epoch invalidates every layer below the guard on every invocation -while leaving the dependency layers above them cached, and the missing-argument -guard means a bare `docker build .` fails loudly instead of quietly reporting a -green it did not earn: an unset build argument is the empty string, which is a -perfectly stable cache key. +question on the host. Its verdict matches the container's: prettier is pinned to +an exact version, installed from `yarn.lock` under `--frozen-lockfile` in both +places, and reads `.gitignore` as its default ignore file — which is why +`.dockerignore` keeps `.gitignore` in the build context. ## Rationale @@ -197,10 +177,9 @@ All work on quak is test-driven. No exceptions. 3. Subsequent commits add the implementation and any refactors needed to make the tests pass. 4. A feature branch can only be merged into `main` when `make check` is green. - `main` is always green. CI runs `script/cibuild`, which lints via - `Dockerfile.lint` and then runs `make test` and `make build` in the - `Dockerfile` image, so neither a red branch nor one that does not compile can - pass CI. + `main` is always green. CI runs `script/cibuild`, which builds the + `Dockerfile`: its `lint` and `test` phases, then the compile, so neither a + red branch nor one that does not compile can pass CI. 5. Tests are the canonical API documentation for this library. Every test file is commented thoroughly enough that a reader who has never seen quak can learn how to use it from the tests alone. Comments explain why a behavior @@ -217,7 +196,7 @@ All work on quak is test-driven. No exceptions. runs `script/lint` — eslint and the prettier check, in the container — but not the tests, and so not the full `make check`. This is deliberate so the TDD red-phase commit (failing tests, no implementation yet) can land. The - suite runs as part of the image build, which is what CI executes via + `test` phase is part of the image build, which is what CI executes via `script/cibuild`, so a red branch still cannot reach `main`. ## Design @@ -235,18 +214,29 @@ quak/ auth/ login flow (SRP + email OTP + TOTP), key unwrap model/ decrypted Collection, File, Metadata types + decrypt fns download/ streaming file/thumbnail download + decryption + library/ the cache-backed Library: metadata store, read + surface, records, content cache, precache, ML data + and search, request pools backup.ts resilient full-account backup with dedup + metadata-backup.ts + backup-metadata: all decrypted metadata as JSON + mldata-fetch.ts fetch + decrypt per-file ML data + filename.ts safe file names from server metadata errors.ts error types shared across layers retry.ts retry classifier + exponential backoff with jitter thumbnails.ts detect + regenerate missing thumbnails client.ts high-level Client class assembled from the above + cli-commands.ts the CLI's commands as functions returning exit codes + cli-output.ts how the CLI prints a file's title and time + cli-read.ts fresh reads for the CLI's read commands + cli-run.ts run a command, print its error, exit with its code + cli-session.ts read the saved session file back into a Client index.ts public library exports bin/ quak.ts CLI entrypoint (commander.js) test/ unit + integration tests (vitest) Makefile - Dockerfile test suite and compile - Dockerfile.lint eslint and prettier, as build steps + Dockerfile lint phase, test phase, compile package.json tsconfig.json ``` @@ -318,11 +308,13 @@ Endpoints used: encrypted token plus key attributes. - `POST /users/ott` and `POST /users/verify-email`: email OTP fallback path. - `POST /users/two-factor/verify`: TOTP second factor. +- `POST /users/logout`: end the calling token's session (`quak logout`). - `GET /collections/v2?sinceTime=`: list collections changed since microsecond timestamp; pass 0 for a full enumeration. - `GET /collections/v2/diff?collectionID=&sinceTime=`: list files in a collection; paginate while `hasMore` is true. - `GET https://files.ente.io/?fileID=`: download encrypted file bytes. +- `POST /files/data/fetch`: fetch encrypted ML data for a batch of files. - `POST /files/upload-url`: mint a presigned upload URL (for thumbnail repair). - `PUT /files/thumbnail`: register an uploaded thumbnail's object key. @@ -362,35 +354,41 @@ and a half seconds of waiting. `sleep` and `random` are injectable through the same option, which is how the test suite exercises the whole policy without waiting. -Two deadlines, applied with `AbortSignal.timeout()` and renewed for each -attempt: +Two deadlines, renewed for each attempt: -| Option | Default | Applies to | -| ------------------- | -------- | ------------------------------------------- | -| `requestTimeoutMs` | `30000` | `getJSON`, `postJSON`, `putJSON`, `putFile` | -| `downloadTimeoutMs` | `600000` | file and thumbnail body transfers | +| Option | Default | Applies to | Kind | +| ------------------- | ------- | ------------------------------------------- | ------------------------------------- | +| `requestTimeoutMs` | `30000` | `getJSON`, `postJSON`, `putJSON`, `putFile` | the whole request | +| `downloadTimeoutMs` | `60000` | file and thumbnail downloads | idle: no bytes received for this long | -They are separate because one number cannot serve both: a value short enough to -keep a hung API call from stalling a backup would cancel a legitimate -multi-gigabyte download. The download deadline covers the body, not just the -headers — `getFileStream` returns as soon as headers arrive, so a deadline that -only guarded the initial request would leave the same hang one layer down. +They are different kinds because a download's length depends on the file and the +link: a whole-transfer deadline short enough to catch a hung connection would +cancel a large video on a slow link that is still making progress. The download +deadline restarts every time bytes arrive, so a slow download runs as long as it +keeps moving, and one that stalls is aborted after 60 seconds of silence. It +covers the wait for the headers and the body — `getFileStream` returns as soon +as headers arrive, so a deadline that only guarded the initial request would +leave the same hang one layer down. There is no limit on the total length of a +download. **Non-idempotent requests are not blindly replayed.** `postJSON` and `putJSON` -reach `/users/srp/create-session`, `/users/two-factor/verify` — which consumes -one of a small number of second-factor attempts — and `/files/thumbnail`. They -are retried only on the three failures that establish no TCP connection to the -server ever existed, so no request byte can have been transmitted: `ENOTFOUND` -and `EAI_AGAIN` (name resolution produced no address) and `ECONNREFUSED` (the -peer refused the connection). A 5xx, a mid-flight reset and a deadline are all -left to the caller, because each of them can happen after the server has already -acted. The routing errnos `EHOSTUNREACH`, `ENETUNREACH` and `ENETDOWN` are -excluded for the same reason, despite looking like connect-time failures: on -Linux an ICMP unreachable arriving mid-flight, or a local interface going down -after the request was written, delivers them on an already-established socket. -They stay retryable for the idempotent calls. `putFile` is exempt: a presigned -PUT stores one whole object at one key in one request, so replaying it has no -partial state to damage. +send every `POST` and `PUT` in the endpoint list above; some of them change +server state, and `/users/two-factor/verify` consumes one of a small number of +second-factor attempts. They are retried only when every errno in the error's +`cause` chain is one of the three that establish no TCP connection to the server +ever existed, so no request byte can have been transmitted: `ENOTFOUND` and +`EAI_AGAIN` (name resolution produced no address) and `ECONNREFUSED` (the peer +refused the connection). A 5xx, a mid-flight reset and a deadline are all left +to the caller, because each of them can happen after the server has already +acted. These two do not follow redirects either: a redirect means the server +already received the request, so it is reported as an error and not retried. The +routing errnos `EHOSTUNREACH`, `ENETUNREACH` and `ENETDOWN` are excluded for the +same reason, despite looking like connect-time failures: on Linux an ICMP +unreachable arriving mid-flight, or a local interface going down after the +request was written, delivers them on an already-established socket. They stay +retryable for the idempotent calls. `putFile` is exempt: a presigned PUT stores +one whole object at one key in one request, so replaying it has no partial state +to damage. A download is retried as a whole — request, stream consumption, and decryption — because a socket reset after the response headers have arrived surfaces in the @@ -422,13 +420,28 @@ decides how to persist sessions. `client.toJSON()` returns a `ClientSnapshot` (a plain serializable object with base64-encoded keys) that the consumer can write to disk, a database, or whatever else fits their use case. `Client.fromJSON(snapshot)` restores a -working client from that snapshot without re-authenticating. +working client from that snapshot without re-authenticating; it checks every +field and each key's length first, and throws an error naming the bad field. +`client.logout()` clears the token and zeroes the key buffers in place; every +later call on that client throws. It does not contact the server, so the token +stays valid there and in any saved snapshot; `await client.logoutOnServer()` +first ends the session on the server (`POST /users/logout`). The CLI stores the snapshot at the platform-appropriate data directory via `env-paths`: `~/Library/Application Support/quak/session.json` on macOS, `$XDG_DATA_HOME/quak/session.json` on Linux. The file is written with mode `0600`. The key material is stored in cleartext in the JSON; treat this file as -you would treat the password itself. +you would treat the password itself. A missing file is reported as "not logged +in"; a file that exists but is corrupt is reported as such, naming the bad +field. Both exit with status 1. + +`quak logout` ends the session on the server, so the token in `session.json` +stops working even in a copy of the file, and then deletes the file. If the +server call fails (or the file is corrupt), the file is still deleted, the +command says the server session could not be ended, and it exits with status 1. +It does not delete the cache: it prints the account's cache directory and says +it still holds decrypted data (file keys in `metadata.json`, cached originals +and thumbnails), for the user to delete if they want it gone. ### CLI surface @@ -436,7 +449,7 @@ you would treat the password itself. quak [--cache-dir ] global: local metadata/content cache location quak login interactive or QUAK_EMAIL/QUAK_PASSWORD quak whoami print logged-in account as JSON -quak logout delete saved session +quak logout end the session, delete it quak collections [--json] list all collections quak files --collection [--json] list files in a collection quak get [--out path] [--collection] download and decrypt a file @@ -448,10 +461,13 @@ quak helper fix-missing-thumbnails [--file ids] generate + upload missing thumbn ``` Every command runs on the same cache-backed library. The read commands — -`collections`, `files`, `get`, and `get-thumb` — force a fresh server round-trip -before they answer, so they report current account state rather than whatever -the cache last held. `--cache-dir` overrides where the cache lives; without it -each account gets its own directory under the per-user cache path. +`collections`, `files`, `get`, `get-thumb`, `backup-metadata`, +`helper list-missing-thumbnails` and `helper fix-missing-thumbnails` — force a +fresh server round-trip before they answer, so they report current account state +rather than whatever the cache last held. If that round-trip fails, the command +prints the error on one line and exits 1. `--cache-dir` overrides where the +cache lives; without it each account gets its own directory under the per-user +cache path. `get` and `get-thumb` resolve the file by ID directly, so `--collection` is accepted for backward compatibility but ignored. `backup-metadata --exif` (alias @@ -459,11 +475,22 @@ accepted for backward compatibility but ignored. `backup-metadata --exif` (alias metadata. The listing and backup commands support `--json` for machine-readable output. +`backup-metadata` fetches ML data in requests of up to 200 files. When a request +still fails after its retries, the error is logged, each of its files is written +with the reason in an `mlDataError` field instead of `mlData`, and the dump goes +on. The exit code is non-zero if any ML data request failed. + `helper fix-missing-thumbnails` regenerates thumbnails for baseline JPEG images only, because the bundled decoder (`jpeg-js`) decodes only JPEG. A non-JPEG image (PNG, HEIC) or a video is reported as `skipped` (unsupported format), kept distinct from a `failed` repair, and does not affect the exit code; a genuine -failure still exits non-zero. +failure still exits non-zero. The server accepts a new thumbnail only from the +file's owner and only when it is no larger than the thumbnail size it records +for the file. So a file another account owns, in an album shared with you, is +skipped by both thumbnail helpers without being fetched, and the fixer skips a +file whose recorded thumbnail size is 0 or unknown. Otherwise the fixer lowers +the quality and size of the thumbnail until it fits, and skips the file if even +the smallest does not. ### Backup layout @@ -478,12 +505,49 @@ failure still exits non-zero. / -> ../../originals/<fileID>.<ext> (symlink) <name>.json collection metadata + file list + failures.json files that failed and have not yet succeeded ``` +`failures.json` records each failed file with the kind of failure, how many +times it has been tried and when it was last tried. A file leaves it once it +succeeds, or once it is no longer in the library or in the backup's scope. The +library's `lib.backup({ includeThumbnails: true })` also writes +`thumbnails/<fileID>.jpg` beside `originals/`; `quak backup` does not. + +A collection's directory and JSON are named after the collection, and a symlink +after the file's title, both with unsafe characters replaced. When two +collections would get the same name, or two files in one collection the same +title (ignoring case in both), each of them gets its ID added: two albums named +`Trip` become `Trip (10)/` and `Trip (11)/`, and two files titled `IMG_0001.JPG` +become `IMG_0001 (12345).JPG` and `IMG_0001 (12346).JPG`. IDs never change, so a +name stays the same from run to run until such a clash appears or goes away. + +Each run removes the symlinks into `originals/` that no longer belong in their +collection's directory, and the directories (and JSON) of collections that were +deleted or renamed. Nothing else in `collections/` is touched: a file or a +symlink you put there stays, and a directory that still holds one after its +symlinks are removed stays too, with its JSON. + Each file is downloaded exactly once regardless of how many collections it -appears in. On subsequent runs, existing originals are skipped. If a download -fails, the error is logged and the backup continues with the next file. The exit -code is non-zero if any files failed. +appears in, and written once: straight into `originals/`, with no copy left in +the cache. An original the cache already held is copied from there instead. On +subsequent runs, existing originals are skipped. If a download fails, the error +is logged and the backup continues with the next file. The exit code is non-zero +if any files failed. `quak backup` opens its library with the thumbnail and +originals precache off, so it fetches only what the backup stores. + +Each original is written to a temporary file in the same directory, synced to +disk, and renamed into place, so an original is either complete or absent, even +after a power cut. A downloaded original's temporary file is named +`.quak-<pid>-<random>.tmp`, one copied from the cache +`.quak-backup-<fileID>.<ext>-<pid>-<random>.tmp`. A run that is killed can leave +one of these temporary files behind; the next backup deletes those whose process +is no longer running. The content cache uses the same scheme, and opening a +library deletes the temporary files in the cache whose process is no longer +running, so a download another process has in progress in the same cache is left +alone. The rename replaces whatever was at the destination rather than writing +through it: a symlink there is replaced, not followed, and the new file has the +temporary file's permissions, not those of the file it replaced. ## TODO @@ -491,7 +555,12 @@ code is non-zero if any files failed. errors - [x] Update the API reference section below to match the current implementation - [x] `make docker` green -- [ ] Tag `v1.0.0` +- [ ] Store live photos in a form a photo viewer can open + (https://git.eeqj.de/sneak/quak/issues/107), once sneak has chosen between + keeping the ZIP and unpacking it + +Tagging and releases are decided by sneak alone, and happen only when he +declares one. Future (desktop client, separate repo): @@ -511,10 +580,11 @@ test suite is the canonical, executable documentation — `test/library/` and ### Opening a library `Library.open(options)` loads the on-disk cache, starts the background refresh -loop, and resolves to a `Library`. On an empty cache it awaits the first refresh -so it never opens onto empty data; on an existing cache it returns immediately -and refreshes in the background, so an unreachable server does not block -opening. +loop, and resolves to a `Library`. On an empty cache it awaits the first +refresh, so it opens onto the account's data whenever the server is reachable; +if that refresh fails, it opens with no data and records the error in +`lib.status()`. On an existing cache it returns immediately and refreshes in the +background, so an unreachable server does not block opening. `LibraryOptions`: @@ -541,7 +611,10 @@ and pass it. The three pools default to 10 / 5 / 25 (see Request pools below). `lib.status()` returns a `LibraryStatus` (collection/file counts, last refresh/ML times and errors, originals usage and effective limit, precache progress, and `closed`). `lib.close()` stops the background timer; it is -idempotent, and an in-flight refresh is left to finish. +idempotent, and an in-flight refresh is left to finish. The promise it returns +resolves once that refresh (including its cache write), the ML data fetch and +the precache fetches already running have all finished, so the cache directory +can then be removed. ### Default reads vs. fresh reads @@ -622,12 +695,15 @@ photos newest first). `lib.subscribe({ onChange })` delivers a `LibraryChange` default limit 20). quak bundles no text encoder, so `searchByEmbedding` takes a query vector the caller produced elsewhere. - `await lib.backup(opts?)` → `BackupResult`. It refreshes, fetches every - in-scope original (and, with `includeThumbnails`, thumbnails) through the - content cache, and rebuilds the on-disk backup tree with a durable failure - ledger. `BackupOptions`: `downloadDirectory` (falls back to the one `open()` - was given), `includeOriginals` (default `true`), `includeThumbnails` (default - `false`), `onlyAlbumNames`, and `onProgress`. See Backup layout above for the - tree it writes. + in-scope original not already in the backup (and, with `includeThumbnails`, + thumbnails) through the content cache, and rebuilds the on-disk backup tree + with a durable failure ledger. A fetched original is written straight into the + backup's `originals/` and not into the cache, which then counts it as present; + one the cache already held is copied from there. `BackupOptions`: + `downloadDirectory` (falls back to the one `open()` was given), + `includeOriginals` (default `true`), `includeThumbnails` (default `false`), + `onlyAlbumNames`, and `onProgress`. See Backup layout above for the tree it + writes. ### Request pools @@ -651,11 +727,19 @@ Under `cacheDirectory`: fetched.json per-file fetch bookkeeping ``` +When `metadata.json` belongs to a different account than the client's, +`Library.open` deletes it and `mldata/` and starts from an empty cache. Cached +originals and thumbnails are kept; they are reached only through the files the +current account's records name. + A stored file appears only via an atomic temp-then-rename, so its presence means -it is complete. The design also calls for a content-hash comparison against -`FileMetadata.hash` on each fetched original; that check is deferred (issue -https://git.eeqj.de/sneak/quak/issues/68) because the exact hash construction -cannot yet be confirmed against the repo's fixtures. +it is complete. Every downloaded original (by `quak get`, the cache, or +`backup`) whose metadata records a content hash (`FileMetadata.hash`) is hashed +as it is written: unkeyed BLAKE2b with a 64-byte output, standard base64. For a +live photo, which is stored as a ZIP, the image and the video are hashed +separately and joined as `<imageHash>:<videoHash>`. A mismatch stores nothing +and fails the download with an error naming the file ID. An original with no +recorded hash, from a very old client, is stored unchecked. ### Key types by source file @@ -704,20 +788,22 @@ documents: commented thoroughly. `main` is always green. - **Required checks before every commit:** `make lint` must pass — that is - eslint plus the prettier check, and it builds `Dockerfile.lint`, so it needs - docker. The pre-commit hook enforces exactly that. `make check` (which also - runs the tests) must pass before merging to `main`. `make fmt-check` is - available for a host-side formatting check on its own, but it is not a - separate requirement: `make lint` already covers it, and running both would - check formatting twice. Never invoke eslint or prettier directly; linting runs - in the container only. + eslint plus the prettier check, and it builds the `lint` phase of the + `Dockerfile`, so it needs docker. The pre-commit hook enforces exactly that. + `make check` (which also runs the tests) must pass before merging to `main`. + `make fmt-check` is available for a host-side formatting check on its own, but + it is not a separate requirement: `make lint` already covers it, and running + both would check formatting twice. Never invoke eslint or prettier directly; + linting runs in the container only. - **Formatting:** prettier with 4-space indents and `proseWrap: always` for markdown. Use `make fmt` to format. Use `yarn` not `npm`. - **Testing:** vitest. Tests go in `test/` mirroring the `src/` structure. - `make test` must complete in under 20 seconds. Use `mkdtempSync` for temporary - directories, never manual timestamp paths. + `make test` must finish in under 60 seconds (the hard cap) and should finish + in under 20. The 90-second `timeout` in the `test` phase of the `Dockerfile` + is a backstop that catches a hung test, not the time limit. Use `mkdtempSync` + for temporary directories, never manual timestamp paths. - **Code style:** `const` for everything, `let` if reassignment is needed, never `var`. Avoid unnecessary comments. No hand-rolled crypto. The diff --git a/REPO_POLICIES.md b/REPO_POLICIES.md index bc2f161..2256291 100644 --- a/REPO_POLICIES.md +++ b/REPO_POLICIES.md @@ -1,6 +1,6 @@ --- title: Repository Policies -last_modified: 2026-07-06 +last_modified: 2026-09-08 --- This document covers repository structure, tooling, and workflow standards. Code @@ -60,17 +60,28 @@ style conventions are in separate documents: prerequisite since nvm requires bash. yarn is then pinned via `corepack prepare yarn@<version> --activate`. Never install "latest" or "lts"; always exact versions. `script/cibuild` runs the CI build: it changes to the - repo root and runs `docker build .`; the Gitea workflow calls it. Four further - scripts are our own extensions to the standard: `script/check` runs - `script/test`, `script/lint`, and `script/fmt-check`; `script/precommit` is - what the git pre-commit hook runs, and it calls `script/check`; - `script/install-precommit` installs the git pre-commit hook (the `make hooks` - target shims to it); and `script/projectname` (literally that filename) simply - outputs the project's name. Scripts that need the name call - `script/projectname` — e.g. `script/docker` assembles its image tag from it — - so those scripts stay byte-identical across all repos. Repo-type-specific - pre-commit extras (e.g. `go mod tidy` verification in Go repos) belong in - `script/precommit`, not in the hook itself. Model scripts are at + repo root, runs `script/bootstrap`, runs `script/check`, and builds the image + with the version; the Gitea workflow calls it. **`script/cibuild` runs + `script/bootstrap` first**, because the workflow checks out the repo and runs + nothing else, while `script/fmt-check` runs the formatter on the host: on a + pristine checkout with nothing installed the run dies there, after the + containerised gates have passed. **The bootstrap alone is not enough**: + `script/bootstrap` installs node and yarn under nvm and leaves neither on the + `PATH` of the shell that called it, so a bare `yarn` still exits 127. The host + entrypoints that need yarn — `script/fmt` and `script/fmt-check` — therefore + source nvm for the pinned node version before invoking it, exactly as + `script/bootstrap`'s own install step does. A runner carrying nothing but + docker and git then gets through `script/check`. Four further scripts are our + own extensions to the standard: `script/check` runs `script/test`, + `script/lint` and `script/fmt-check`; `script/precommit` is what the git + pre-commit hook runs, and it calls `script/check`; `script/install-precommit` + installs the git pre-commit hook (the `make hooks` target shims to it); and + `script/projectname` (literally that filename) simply outputs the project's + name. Scripts that need the name call `script/projectname` — e.g. + `script/docker` assembles its image tag from it — so those scripts stay + byte-identical across all repos. Repo-type-specific pre-commit extras (e.g. + `go mod tidy` verification in Go repos) belong in `script/precommit`, not in + the hook itself. Model scripts are at `https://git.eeqj.de/sneak/prompts/raw/branch/main/script/<name>`. The README must document the provided scripts in an **Entrypoints** section (see the README requirements below). @@ -89,87 +100,140 @@ style conventions are in separate documents: contributor should be able to understand the entire development workflow by reading the Makefile. -- Every repo should have a `Dockerfile`. All Dockerfiles must run `make check` - as a build step so the build fails if the branch is not green. For non-server - repos, the Dockerfile should bring up a development environment and run - `make check`. For server repos, `make check` should run as an early build - stage before the final image is assembled. Dockerfiles install development - prerequisites by running `script/bootstrap` rather than duplicating installs - inline; COPY `script/` and the dependency manifests (`package.json` + - `yarn.lock`, `go.mod` + `go.sum`, etc.) before running it so the bootstrap - layer stays cached until dependencies change. +- Every repo should have a `Dockerfile`, and it carries the repo's gates: a + `lint` phase and a `test` phase, with the final stage depending on both so the + image cannot be built unless they pass. For non-server repos the final stage + brings up a development environment; for server repos it is the runtime image. + Dockerfiles install development prerequisites by running `script/bootstrap` + rather than duplicating installs inline; COPY `script/` and the dependency + manifests (`package.json` + `yarn.lock`, `go.mod` + `go.sum`, etc.) before + running it. -- **Dockerfiles must use a separate lint stage for fail-fast feedback.** Go - repos use a multistage build where linting runs in an independent stage based - on the `golangci/golangci-lint` image (pinned by hash). This stage runs - `make fmt-check` and `make lint` before the full build begins. The build stage - then declares an explicit dependency on the lint stage via - `COPY --from=lint /src/go.sum /dev/null`, which forces BuildKit to complete - linting before proceeding to compilation and tests. This ensures lint failures - surface in seconds rather than minutes, without blocking on dependency - download or compilation in the build stage. +- **Linting and testing run in Docker, as phases of the `Dockerfile`.** There is + no separate lint file. `script/lint` and `script/test` each build one phase + and nothing else: - The standard pattern for a Go repo Dockerfile is: + ```sh + docker build --no-cache --target lint -t "$(script/projectname)-lint" . + docker build --no-cache --target test -t "$(script/projectname)-test" . + ``` + + **A stage that is not the last one in the file is built only when the final + stage's chain depends on it, or when `--target` names it.** That is why the + two gates are always invoked by name here, and why the final stage carries a + `COPY --from=` of a harmless file from each of them: without that edge a + plain `docker build .` builds the last stage alone and exits 0 having linted + and tested nothing. + + **Every `docker build` in `script/` is tagged**, here and in + `script/cibuild` and `script/docker`. An untagged build leaves a dangling + image behind on every invocation, on every developer host and every CI + runner; a tagged one replaces the previous image. + + Inside a phase the tool is invoked directly — `golangci-lint`, `go test`, + `eslint`, `prettier` — never through `make lint` or `script/test`, which are + themselves a `docker build` and would recurse into a daemon that does not + exist in a build step. Formatting is the exception and stays on the host: + `script/fmt` writes the working tree, and `script/fmt-check` is its + read-only twin. + + **No lint verdict may come from a host invocation of the linter.** On a + shared host golangci-lint reads a result cache keyed on file content rather + than location, so a second checkout of the same content is served the first + one's findings, and a host-global lock in `$TMPDIR` makes concurrent runs + exit non-zero with `parallel golangci-lint is running` — a status a caller + cannot tell from real findings. Both have produced wrong verdicts in this + org, in both directions. A container has its own cache, its own `TMPDIR` and + a digest-pinned binary, so neither is reachable. + +- **Any build that runs checks is built with `--no-cache`.** Docker invalidates + a `COPY` layer only when the copied content changes, so on an unchanged tree + the check `RUN` is served from cache, nothing executes, and the build still + exits 0. Every `docker build` in `script/` therefore passes `--no-cache`: + `script/lint`, `script/test`, `script/cibuild` and `script/docker` are the + four, and there is no fifth — `script/check` runs the two gate phases and + `script/fmt-check`, and builds no image of its own. A bare `docker build .` is + not evidence that anything ran: a sub-second build reporting success is a + cache hit, not a result. Never invalidate by pruning — `docker builder prune` + and friends destroy a build cache shared with every other build on the host. + +- **The gate phases are separate stages, and the build stage depends on both.** + The lint phase is based on the `golangci/golangci-lint` image (pinned by + hash), so lint failures surface in seconds rather than after a full compile, + and the test phase is based on the Go image. The canonical Go repo + `Dockerfile`: ```dockerfile - # Lint stage — fast feedback on formatting and lint issues + # Lint phase # golangci/golangci-lint:v2.x.x, YYYY-MM-DD FROM golangci/golangci-lint@sha256:... AS lint WORKDIR /src COPY go.mod go.sum ./ RUN go mod download COPY . . - RUN make fmt-check - RUN make lint + RUN golangci-lint run --config .golangci.yml ./... - # Build stage + # Test phase # golang:1.x-alpine, YYYY-MM-DD + FROM golang@sha256:... AS test + WORKDIR /src + COPY go.mod go.sum ./ + RUN go mod download + COPY . . + RUN go test -timeout 90s -race -cover ./... || \ + { echo "--- Rerunning with -v for details ---"; \ + go test -timeout 90s -race -v ./...; exit 1; } + + # Build stage. Nothing is wanted from either phase above; the copies + # are what make BuildKit build them first, so this stage cannot run + # unless lint and test passed. + # golang:1.x-alpine, YYYY-MM-DD FROM golang@sha256:... AS builder + COPY --from=lint /src/go.sum /dev/null + COPY --from=test /src/go.sum /dev/null WORKDIR /src - - # Force BuildKit to run the lint stage before proceeding - COPY --from=lint /src/go.sum /dev/null - COPY go.mod go.sum ./ RUN go mod download COPY . . - RUN make test ARG VERSION=dev RUN CGO_ENABLED=0 go build -trimpath \ -ldflags="-s -w -X main.Version=${VERSION}" \ -o /app ./cmd/app/ - # Runtime stage + # Runtime stage, and the last one FROM alpine@sha256:... COPY --from=builder /app /usr/local/bin/app ENTRYPOINT ["app"] ``` Key points: - - The lint stage uses the `golangci/golangci-lint` image directly (it - includes both Go and the linter), so there is no need to install the - linter separately. - - `COPY --from=lint /src/go.sum /dev/null` is a no-op file copy that creates - a stage dependency. BuildKit runs stages in parallel by default; without - this line, the build stage would not wait for lint to finish and a lint - failure might not fail the overall build. + - The lint phase uses the `golangci/golangci-lint` image directly (it has + both Go and the linter), so nothing needs installing. + - `COPY --from=<phase> /src/go.sum /dev/null` is a no-op copy whose only + purpose is the ordering edge. BuildKit runs stages in parallel by default, + and a stage nothing depends on is not built at all, so without these two + lines a red gate would not fail the build. + - Keep the runtime stage last, and if you add a stage after it, give it the + same two copies. A plain `docker build .` builds the last stage's chain + and nothing else. - If the project uses `//go:embed` directives that reference build artifacts - (e.g. a web frontend compiled in a separate stage), the lint stage must + (e.g. a web frontend compiled in a separate stage), the lint phase must create placeholder files so the embed directives resolve. Example: `RUN mkdir -p web/dist && touch web/dist/index.html web/dist/style.css`. - The lint stage should not depend on the actual build output — it exists to - fail fast. - If the project requires CGO or system libraries for linting (e.g. - `vips-dev`), install them in the lint stage with `apk add`. - - The build stage runs `make test` after compilation setup. Tests run in the - build stage, not the lint stage, because they may require compiled - artifacts or heavier dependencies. + `vips-dev`), install them in the lint phase with `apk add`. + - `ARG VERSION=dev` is declared in the stage that compiles and supplied by + `script/docker` and `script/cibuild`; no stage may call `git describe`. - Every repo should have a Gitea Actions workflow (`.gitea/workflows/`) that - runs `script/cibuild` (which runs `docker build .`) on push. Since the - Dockerfile already runs `make check`, a successful build implies all checks - pass. + runs `script/cibuild` on push, and checks out the repo as its only other step. + That script bootstraps, runs the gate phases, and then builds the image, so a + successful run means every check passed; a bare `docker build .` does not + carry the same guarantee, because its gate phases may come from the cache. The + image build is uncached and so runs the gate phases a second time. That is the + price of the rule above, and it is worth paying: the image that ships is built + from a run of its own gates rather than from a cache entry. - Use platform-standard formatters: `black` for Python, `prettier` for JS/CSS/Markdown/HTML, `go fmt` for Go. Always use default configuration with @@ -189,14 +253,21 @@ style conventions are in separate documents: module under test to verify it compiles/parses. There is no excuse for `make test` to be a no-op. -- `make test` must complete in under 20 seconds. Add a 30-second timeout in the - Makefile. +- `make test` must complete in under 60 seconds. That is the hard cap, and a + suite that exceeds it fails. Under 20 seconds is the target. A suite between + 20 and 60 seconds is still green, but the overage must be filed as an + improvement bug against that repo. Add a 90-second timeout to the test + invocation (`go test -timeout 90s`). The backstop deliberately sits above the + hard cap so that it catches a genuinely hung test rather than a merely slow + one. -- **`make test` should use the conditional verbose rerun pattern.** Run tests - without `-v` (verbose) first. If tests fail, automatically rerun with `-v` to - show full output. This keeps CI logs and `docker build` output clean on - success (just package/suite summaries) while providing full diagnostic detail - on failure (every test case, every assertion). The general shell pattern: +- **The test command should use the conditional verbose rerun pattern.** Run + tests without `-v` (verbose) first. If tests fail, automatically rerun with + `-v` to show full output. This keeps CI logs and `docker build` output clean + on success (just package/suite summaries) while providing full diagnostic + detail on failure (every test case, every assertion). The command lives in the + `test` phase of the `Dockerfile`, since `script/test` builds that phase; the + Makefile form below is the same pattern for any repo-local invocation: ```makefile test: @@ -209,11 +280,24 @@ style conventions are in separate documents: ```makefile test: - @go test -timeout 30s -race -cover ./... || \ + @go test -count=1 -timeout 90s -race -cover ./... || \ { echo "--- Rerunning with -v for details ---"; \ - go test -timeout 30s -race -v ./...; exit 1; } + go test -count=1 -timeout 90s -race -v ./...; exit 1; } ``` + `-count=1` is required on both invocations: it defeats Go's test _result_ + cache, so the target cannot report a pass it did not earn, and the rerun + reproduces a failure instead of replaying it. It leaves the build cache + alone, so it costs the runtime of the suite and no recompilation. + + Note that this is a second, independent cache, stacked below the Docker + layer cache that [issue #26](https://git.eeqj.de/sneak/prompts/issues/26) + addresses. `CHECK_EPOCH` guarantees the `RUN make test` _step_ re-executes; + it does not guarantee `go test` inside that step does any work, because the + `GOCACHE` baked into earlier image layers survives into the re-executed + step. They are two separate defects requiring two separate fixes, and a fix + for one must not be recorded as covering the other. + Python example: ```makefile @@ -239,10 +323,83 @@ style conventions are in separate documents: must be in `.gitignore`. No exceptions. - `.gitignore` should be comprehensive from the start: OS files (`.DS_Store`), - editor files (`.swp`, `*~`), language build artifacts, and `node_modules/`. - Fetch the standard `.gitignore` from - `https://git.eeqj.de/sneak/prompts/raw/branch/main/.gitignore` when setting up - a new repo. + editor files (`.swp`, `*~`), in-repo agent scratch directories (`.claude/`), + language build artifacts, and `node_modules/`. Fetch the standard `.gitignore` + from `https://git.eeqj.de/sneak/prompts/raw/branch/main/.gitignore` when + setting up a new repo. These patterns are written to `.gitignore`'s own + semantics, in which an unanchored pattern already matches at every depth; they + are not a `.dockerignore` and must not be transplanted into one unmodified. + +- **`.dockerignore` does not use `.gitignore` semantics, and copying patterns + across unmodified leaves secrets in the build context.** Docker matches with + `moby/patternmatcher`: `filepath.Match` semantics plus a `**` extension, so + `*` does not cross `/` and a pattern without a leading `**/` is anchored at + the build-context root. A `.dockerignore` listing `.env`, `*.pem` and `*.key` + therefore excludes only the copies at the repository root, while `config/.env` + and `certs/server.key` still reach the context and can land in an image layer + — which is more dangerous than a short file with no secret patterns at all, + because it reads as solved and stops anyone looking. Give every + depth-independent pattern the `**/` prefix and leave only genuinely + root-anchored entries unprefixed: `.git`, and the repo's own host-built + binary, written `/myapp` and never `**/myapp`, which would also match + `cmd/myapp/` and delete the package directory from the context. Matching is + case-sensitive, and an ALL-CAPS twin per pattern still misses `Server.Key`, so + secret names use character ranges — `**/*.[kK][eE][yY]`, `**/*.[pP][eE][mM]`, + and likewise for `.envrc` and the extensionless SSH keys. Where such a pattern + also catches something the build needs, re-include it with a negation + (`!docs/example.env`); deleting the pattern reopens the exposure for every + other file it covers. Fetch the standard `.dockerignore` from + `https://git.eeqj.de/sneak/prompts/raw/branch/main/.dockerignore` and extend + it with the repo's own artifacts. + +- **In-repo agent scratch belongs in both files, written to each file's own + semantics.** `.claude/` holds one worktree per in-flight agent — an entire + additional checkout of the repo — so under `COPY . .` the build context + inflates by a multiple of the repo and another session's unreviewed work can + be copied into an image layer. In `.gitignore` the entry is `.claude/`, + unanchored. In `.dockerignore` it is `.claude`, anchored and with **no** `**/` + prefix, because the prefixed form would also delete any nested directory of + that name from the build. Anchoring carries a known gap that the canonical + `.dockerignore` states in its own comment, since consuming repos receive the + file and not the tracker: the directory is created in the agent's working + directory, so a repo running agents in subdirectories still ships + `services/api/.claude/` and must add its own anchored entry there. + +- **Excluding `.git` means `git describe` cannot run inside any build stage, and + it fails quietly there.** In a build stage there is no repository, so + `git describe` writes nothing to stdout, `-X main.Version=` comes out empty, + the binary reports no version at all, and the build still exits 0. Compute the + version on the host and thread it in as a build arg. `script/docker` and + `script/cibuild` do this, byte-identically across repos: + + ```sh + # Own line: a failing command substitution inside an argument does not + # trip `set -e`, so the inline form degrades to an empty constant. + version="$(git describe --tags --always --dirty 2>/dev/null || true)" + [ -n "$version" ] || version="unknown" + docker build --no-cache \ + --build-arg VERSION="$version" \ + -t "$(script/projectname)" . + ``` + + `--always` makes an untagged repo yield an abbreviated commit hash rather + than failing, and the `[ -n "$version" ]` line is the single place the + fallback is applied — a live check that fires on a build from an export with + no `.git` and on a repository with no commits yet. Do not fold it into the + substitution as `|| echo unknown`, which makes the guard unreachable. The + Dockerfile's side is `ARG VERSION=dev` in the stage that compiles, declared + there because `ARG` is stage-scoped; passing `VERSION` to a repo whose + Dockerfile declares no such `ARG` is ignored and costs nothing, which is why + the scripts stay byte-identical. One consequence for CI: the standard + checkout action clones shallow and fetches no tags, so a repo that embeds a + tag-derived version must set `fetch-depth: 0` on its checkout step. + +- **Verify `.dockerignore` by enumerating the image, not by reading the + patterns.** Plant files at the root _and_ at least two directories deep, build + a probe image that does `COPY . .`, and list what actually landed + (`docker run --rm --entrypoint find IMAGE /app`). The `transferring context` + size is not a substitute: a nested secret is a few bytes, and BuildKit + transfers only the delta from the previous build. - **No build artifacts in version control.** Code-derived data (compiled bundles, minified output, generated assets) must never be committed to the @@ -258,9 +415,45 @@ style conventions are in separate documents: - Make all changes on a feature branch. You can do whatever you want on a feature branch. -- `.golangci.yml` is standardized and must _NEVER_ be modified by an agent, only - manually by the user. Fetch from - `https://git.eeqj.de/sneak/prompts/raw/branch/main/.golangci.yml`. +- `.golangci.yml` is standardized. The vendored copy in a consuming repo must + _NEVER_ be modified by an agent: fetch it from + `https://git.eeqj.de/sneak/prompts/raw/branch/main/.golangci.yml` and keep it + byte-identical, so that no repo can quietly loosen its own linting. Linter + configuration changes are made to the canonical copy in the `prompts` repo and + reach consuming repos by re-vendoring; an agent may open a PR against + canonical, which only the user merges. One list is exempt from byte-identity, + because it cannot be written once for every repo: the `deny` list of the + `test-support` depguard rule, where a repo names its own test-support packages + by full import path. A repo adds entries there and changes nothing else, and a + re-vendor carries its entries forward. The canonical golangci-lint version is + v2.12.2 (released 2026-05-06), pinned as the digest of the lint phase's base + image + (`golangci/golangci-lint@sha256:5cceeef04e53efe1470638d4b4b4f5ceefd574955ab3941b2d9a68a8c9ad5240`, + which reports `2.12.2 built with go1.26.2 from c0d3ddc9`). That digest is the + only pin, since no repo installs golangci-lint on the host: bumping the + version means changing it and nothing else. + +- **`script/bootstrap` installs a pinned tool by comparing versions, never by + testing presence.** An `if ! command -v <tool>; then install; fi` guard tests + `PATH` only, so on an already-provisioned machine the pin is inert and a + version bump is a silent no-op — while the Dockerfile, installing into a clean + image, gets the pinned version, so a local `make check` and `make docker` can + disagree about what the tool even is. The canonical form: + - compares the installed version against the pin over the **whole** version + token; a parser that stops at the first `-` reports `2.12.2` for a host + running `2.12.2-rc1` and skips the install; + - treats absent, non-zero, empty or unrecognised `--version` output as a + mismatch, so the failure direction is a redundant install and never a + skipped one; + - after installing, re-resolves the binary the way callers do — `hash -r`, + then through `PATH`, not through the directory the installer wrote to — + and fails naming the resolved path, since an install that a shadowing + binary hides succeeds while changing nothing any caller sees; + - is actually called, and prints the version on both success paths: a + function defined and never invoked has the same exit status and the same + empty output as one that worked. + + Keep it POSIX sh: no arrays, no `[[`, no `grep -P`. - When pinning images or packages by hash, add a comment above the reference with the version and date (YYYY-MM-DD). @@ -379,7 +572,9 @@ style conventions are in separate documents: language-specific config). Everything else goes in a subdirectory. Canonical subdirectory names: - `bin/` — executable scripts and tools - - `cmd/` — Go command entrypoints + - `cmd/` — Go command entrypoints; thin only: one `main.go` per binary whose + body is a single call into `internal/` or `pkg/`, no project logic in + `cmd/` - `configs/` — configuration templates and examples - `deploy/` — deployment manifests (k8s, compose, terraform) - `docs/` — documentation and markdown (README.md stays in root) diff --git a/TODO.md b/TODO.md index 2300972..35b9e6f 100644 --- a/TODO.md +++ b/TODO.md @@ -14,10 +14,214 @@ pre-1.0 # Next Step -Tag v1.0.0. +Store live photos in a form a photo viewer can open +(https://git.eeqj.de/sneak/quak/issues/107). This waits on sneak's choice +between keeping the ZIP and unpacking it into the image and the video. + +Tagging and releases are decided by sneak alone, and happen only when he +declares one. # Completed Steps +- 2026-09-23: Settled the package metadata (issue 6). quak is not published, so + `package.json` is marked `"private": true` and the `files` field is gone. + `engines.node` is `>=22`, the major version `script/bootstrap` and the + `Dockerfile` use. An `exports` map makes `.` and `./package.json` the only + importable paths; `runMetadataBackup`, the thumbnail helpers and their types + stay internal to the CLI. + +- 2026-09-23: Brought the README and this file in line with the tree (issue + 111). The layout lists `src/library/` and the other source files, the backup + layout names `failures.json` and the optional `thumbnails/`, "Opening a + library" says what happens when the first refresh fails, the Testing section + gives the 60-second hard cap and 20-second target for `make test` and names + the 90-second `timeout` in the `Dockerfile` as the backstop for a hung test, + and "Tag v1.0.0" is no longer listed as the next step. + +- 2026-09-23: Tested `quak login` and `backup-metadata --exif` (issue 110). + `loginCommand` takes its login function and its prompts from `CliContext`, and + `bin/quak.ts` passes `Client.login` and the terminal prompts. Tests cover a + login from `QUAK_EMAIL` and `QUAK_PASSWORD` with no prompt, the TOTP prompt, a + failed login, and the saved session's modes, and show that `--exif` and + `--all` each turn on EXIF extraction and that it is off without them. + +- 2026-09-23: `quak backup` writes each original once and no longer fills the + cache (issue 106). An original fetched for a backup is written by the download + writer straight into the backup's `originals/`, and the content cache records + it there instead of keeping its own copy; one the cache already held is still + copied. `quak backup` opens its library with the thumbnail and originals + precache off. + +- 2026-09-23: `backup-metadata`, `helper list-missing-thumbnails` and + `helper fix-missing-thumbnails` refresh before they answer (issue 100). Each + awaits `lib.fresh()` before reading, so a file added since the cache was + written is included, and a failed refresh prints one line and exits 1 instead + of answering from a stale or empty cache. The README lists them among the + commands that refresh first. + +- 2026-09-23: `quak backup` waits for the server refresh and fails when it fails + (issue 99). `lib.backup()` joins a refresh already running or starts one, as + `fresh()` does, and rejects before touching any file when it fails, leaving + `failures.json` as it was, so `quak backup` prints the error as one line and + exits 1 instead of backing up the previous run's file list, or nothing, and + exiting 0. + +- 2026-09-23: CLI errors print a message instead of a stack trace (issue 102). + An error a command throws is printed as one `quak: MESSAGE` line on stderr and + the CLI exits 1 once output has drained. The wrapper that does this moved from + `bin/quak.ts` to `src/cli-run.ts`, and `bin/quak.ts` now awaits + `program.parseAsync()`. + +- 2026-09-23: Opening a library no longer deletes another process's download in + progress (issue 105). The download writer's temp files are named + `.quak-<pid>-<random>.tmp`, and `removeLeftoverTempFiles`, moved from the + backup into the download module, deletes a `.quak-*.tmp` file only when the + process ID in its name is no longer running. The content cache calls it at + `open()` for `originals/` and `thumbnails/`, the backup as before. + +- 2026-09-23: Re-vendored the lint and test setup from the template (issue 96). + Linting and testing are the `lint` and `test` phases of the `Dockerfile`; + `script/lint` and `script/test` each build one with `--no-cache`, and the last + stage compiles and depends on both, so `script/cibuild` is one build. + `Dockerfile.lint`, `CHECK_EPOCH`, `LINT_EPOCH` and the tests that checked them + are gone; `REPO_POLICIES.md` is re-copied. + +- 2026-09-23: Stopped `helper fix-missing-thumbnails` retrying files the server + always refuses (issue 109). Both thumbnail helpers skip a file another account + owns without fetching it. The fixer skips a file whose recorded thumbnail size + is 0 or unknown before downloading it, and otherwise tries smaller encodings + (720 px quality 50 down to 160 px quality 20) until the encrypted thumbnail is + no larger than that size, skipping the file if none fits. + +- 2026-09-23: Tested the live-photo hash check's error paths (issue 117). Tests + download a live photo whose ZIP names an unknown compression method, one whose + ZIP has no image entry and one with no video entry, and check that nothing is + stored and the error names the file ID; the unreadable one is not retried. + +- 2026-09-23: `quak logout` ends the session on the server (issue 108). It calls + `POST /users/logout` through the new `Client.logoutOnServer()`, then deletes + `session.json` even when that call fails, says so and exits 1. It prints the + account's cache directory and says it still holds decrypted data. The default + cache path is now `defaultCacheDirectory()` in the library, shared with + `Library.open`. + +- 2026-09-23: Fixed the backup's per-collection folders (issue 103). Two files + in one collection with the same title, and two collections with the same name, + each get their ID added to the name (`IMG_0001 (12345).JPG`, `Trip (10)/`), so + none replaces another's symlink or JSON. Each run removes symlinks into + `originals/` for files no longer in the collection, and the folders of deleted + or renamed collections, leaving anything else in `collections/` alone. The + README backup layout states the naming rule. + +- 2026-09-23: Checked downloaded originals against their recorded content hash + (issue 68). `downloadFile`, which `quak get`, the content cache and backup all + use, hashes the decrypted bytes (unkeyed BLAKE2b-512, standard base64) and + stores nothing on a mismatch, failing with an error naming the file ID. A live + photo ZIP is unpacked as it streams with `fflate` and its image and video + hashed separately as `<imageHash>:<videoHash>`. `decryptFile` reads older + clients' `imageHash` and `videoHash` fields for live photos. A file with no + recorded hash is stored unchecked. + +- 2026-09-23: Kept one account's cache from mixing with another's (issue 104). + When `metadata.json` in the cache directory was written for a different, + non-zero user ID than the client's, `Library.open` deletes it and `mldata/` + and starts empty, so the first refresh enumerates from 0. This only happens + with `--cache-dir` or an explicit `cacheDirectory`; the default path already + includes the user ID. A test opens one account's cache as another account. + +- 2026-09-23: `backup-metadata` no longer stops on one failed ML data request + (issue 101). Each request of up to 200 files is tried on its own; a failed one + is logged, its files are written with the reason in `mlDataError`, and the + command exits 1 once the dump is complete. `fetchMLData`, which only this + command used, is gone; the command calls `fetchMLDataBatch` per batch. +- 2026-09-23: Single-sourced the version string (issue 5). `package.json` is the + only place it is written: `src/index.ts` imports it for `VERSION` and + `bin/quak.ts` passes `VERSION` to commander. tsc copies `package.json` to + `dist/package.json`, so the import resolves from the built output too, and + `script/build` runs the built CLI with `--version` to prove it. A test checks + that `VERSION` and `quak --version` both equal the `package.json` version. +- 2026-09-23: Tested that `Library.close()` waits for the originals precache + (issue 93). The test that holds a precache fetch open while `close()` runs now + runs once with only the thumbnail fill and once with only the originals fill, + so dropping either wait from `Precache.close()` fails a test. +- 2026-09-23: Fixed two intermittently failing library tests (issue 90). + `Library.close()` now returns a promise that resolves once an in-flight + refresh (including its cache write), the ML data fetch and running precache + sweeps have finished; the library tests await it, so `afterEach` no longer + removes the cache directory while something is still writing into it. The + precache test waits for both fills to report "done" instead of for its stub + source to be called, which happened before the cache recorded the file. +- 2026-09-23: Pinned three guards reviewers found untested (issue 89). The + download idle deadline's timer is unref'd, so it can never keep the process + alive, and a test checks no timer is left after a download completes or fails. + A test covers the rejection of `#` in a request path. The EXIF scan compares + the `Exif` header only in an APP1 segment of length 8 or more, so it never + reads the next segment's bytes, with a test for a short one. +- 2026-09-23: Made the download deadline an idle deadline (issue 24). + `downloadTimeoutMs` now aborts a file or thumbnail download only after no + bytes have arrived for that long, default 60 seconds, instead of bounding the + whole transfer at 10 minutes, so a slow download that keeps making progress + completes. A download that fails before reading the whole body cancels it, so + a failed file no longer holds its connection. +- 2026-09-23: Every `ApiClient` request URL is now built by one function next to + the class (issue 18), so a self-hosted `apiOrigin` with a base path keeps it + on every request, a path works with or without a leading slash, and query + parameters are percent-encoded. A path containing `?` or `#` is rejected with + an error instead of being silently cut. +- 2026-09-23: Made the CLI testable and tested it (issue 12). The command bodies + moved from `bin/quak.ts` into `src/cli-commands.ts` as functions that take + their options and a context (output streams, session directory, cache + directory, session loader) and return an exit code; `bin/quak.ts` only wires + them to commander and exits with the code once stdout and stderr have drained, + so nothing below it calls `process.exit`. `test/cli/commands.test.ts` drives + them with a fake client: session file modes, logout, the missing and corrupt + session paths, and the output and exit code of `whoami`, `collections`, + `files`, `get`, `get-thumb`, `backup` and `helper list-missing-thumbnails`. +- 2026-09-23: Hardened the backup tree's atomic copy (issue 22). `copyAtomic` + fsyncs its temp file before the rename and the directory after it, through the + download writer's `fsyncPath`; each backup run deletes `.quak-backup-*.tmp` + files whose process is no longer running. The README backup layout names the + temp files and states that the rename replaces a symlink and takes the temp + file's permissions. Added tests for a missing and an unwritable destination + directory for `downloadFile` and `downloadThumbnail`. +- 2026-09-23: Hardened the JPEG EXIF scan behind `backup-metadata --exif` (issue + 11). Every segment length is checked against the remaining bytes and lengths + under 2 stop the scan, so a truncated or corrupt original can neither throw + nor loop. A malformed or unparseable EXIF segment is recorded as + `imageMetadata.exifError`, and a failure to read the original as + `imageMetadataError` in the per-file JSON, instead of the field being left + out. +- 2026-09-22: Hardened the retry classifier (issue 80). A `POST` or `PUT` is + replayed only when every errno in the cause chain is a connect errno, and it + no longer follows redirects. `getRetryOptions()` returns a copy. Tests pin + every errno the classifier names, the cause-chain depth limit, cycle + termination, and a fresh deadline per attempt for every retrying entry point. + The README's endpoint list is the one place that names the requests the replay + rule covers. +- 2026-09-22: Stopped `make test` collecting tests from checkouts nested under + `.claude/` (issue 25). vitest ignores `.gitignore` when finding tests, so a + nested checkout ran the whole suite again; `vitest.config.ts` now adds + `.claude/**` to vitest's default excludes, and + `test/packaging/nested-checkout.test.ts` plants a nested checkout in a temp + directory and fails if vitest would collect it. +- 2026-09-22: Dropped the deprecated `@types/libsodium-wrappers-sumo` stub from + `devDependencies` (issue 27). It shipped no declarations; the types come from + `libsodium-wrappers-sumo` itself. `yarn.lock` regenerated by `yarn remove`. +- 2026-09-22: Hardened the client session lifecycle (issue 10). + `Client.fromJSON` checks every snapshot field and each key's decoded length + and names the bad field; `toJSON` reads the token through + `ApiClient.getAuthToken` and throws when there is none; `logout` zeroes the + key buffers, and `collectionsSince` re-checks for logout after its request so + it never decrypts with zeroed keys. The CLI reports a corrupt session file + separately from a missing one (`src/cli-session.ts`). +- 2026-09-22: Sanitized file names taken from server metadata (issue 9). A new + `src/filename.ts` holds the one sanitizer, used by `quak get`/`get-thumb` + without `--out`, `downloadFile`/`downloadThumbnail` without `outPath`, and the + backup and metadata backup trees; it removes separators, control characters, + leading dots and Windows device names, and falls back to a name built from the + ID for an empty title. Originals-cache extensions are letters and digits only, + else `.bin`. A user-supplied path is used as is. `decryptFile` reads a missing + or non-string title as "" and rejects metadata that is not a JSON object. - 2026-09-22: Rewrote the README API reference (and the Getting Started / usage snippets) to match the shipped cache/API library on `next` (issue 53, issue 13). Documented `Library.open` and its options, the default-read vs `fresh()` diff --git a/bin/quak.ts b/bin/quak.ts index 068620c..4c02183 100644 --- a/bin/quak.ts +++ b/bin/quak.ts @@ -1,218 +1,79 @@ #!/usr/bin/env node -import { input, password as passwordPrompt } from "@inquirer/prompts"; import { stdout, stderr } from "node:process"; -import { - copyFileSync, - existsSync, - mkdirSync, - readFileSync, - writeFileSync, -} from "node:fs"; -import { join } from "node:path"; +import { input, password } from "@inquirer/prompts"; import { Command } from "commander"; import envPaths from "env-paths"; -import { Client, type ClientSnapshot } from "../src/client.js"; import { init } from "../src/crypto/index.js"; -import { Library, type LibraryClient } from "../src/library/index.js"; import { - fileListRow, - fileListLine, - originalName, - thumbnailName, -} from "../src/cli-output.js"; -import { freshCollections, freshFiles, freshFile } from "../src/cli-read.js"; -import { runMetadataBackup } from "../src/metadata-backup.js"; -import { - listMissingThumbnails, - fixMissingThumbnails, -} from "../src/thumbnails.js"; + type CliContext, + loginCommand, + whoamiCommand, + logoutCommand, + collectionsCommand, + filesCommand, + getCommand, + getThumbCommand, + backupMetadataCommand, + backupCommand, + listMissingThumbnailsCommand, + fixMissingThumbnailsCommand, +} from "../src/cli-commands.js"; +import { run as runCommand } from "../src/cli-run.js"; +import { loadSession } from "../src/cli-session.js"; +import { Client } from "../src/client.js"; +import { VERSION } from "../src/index.js"; const paths = envPaths("quak", { suffix: "" }); -const sessionPath = join(paths.data, "session.json"); - -const loadSession = (): ClientSnapshot | null => { - if (!existsSync(sessionPath)) return null; - try { - return JSON.parse(readFileSync(sessionPath, "utf-8")) as ClientSnapshot; - } catch { - return null; - } -}; - -const saveSession = (snapshot: ClientSnapshot): void => { - mkdirSync(paths.data, { recursive: true, mode: 0o700 }); - writeFileSync(sessionPath, JSON.stringify(snapshot, null, 2), { - mode: 0o600, - }); -}; - -const requireSession = (): Client => { - const snapshot = loadSession(); - if (!snapshot) { - stderr.write( - `Not logged in. Run "quak login" first.\nSession file: ${sessionPath}\n`, - ); - process.exit(1); - } - return Client.fromJSON(snapshot); -}; - -const prompt = async (message: string): Promise<string> => input({ message }); - -const promptSecret = async (message: string): Promise<string> => - passwordPrompt({ message, mask: true }); const program = new Command(); program .name("quak") .description("CLI for the Ente end-to-end encrypted photo service") - .version("0.0.0") + .version(VERSION) .option( "--cache-dir <path>", "Directory for the local metadata/content cache " + "(default: the per-user cache directory)", ); -// The `--cache-dir` global, or undefined to let the library pick its per-user -// default keyed by the account id. -const cacheDirOption = (): string | undefined => - program.opts<{ cacheDir?: string }>().cacheDir; - -// A library client that omits `fetchMLData`, so the point commands below do not -// kick the library's background ML backfill: they read metadata, or fetch one -// file's content, and exit. `backup` and `backup-metadata` handle ML on their -// own terms. The content source is kept so `get`/`get-thumb`/`--exif` can fetch -// originals through the on-disk cache. -const readLibraryClient = (client: Client): LibraryClient => ({ - whoami: () => client.whoami(), - collectionsSince: (args) => client.collectionsSince(args), - filesSince: (args) => client.filesSince(args), - contentSource: () => client.contentSource(), +const context = (): CliContext => ({ + stdout, + stderr, + sessionDir: paths.data, + cacheDir: program.opts<{ cacheDir?: string }>().cacheDir, + loadSession, + login: (opts) => Client.login(opts), + prompt: (message) => input({ message }), + promptSecret: (message) => password({ message, mask: true }), }); -// Open a library for a single point command: the aggressive background precache -// (issue #48) is off — a one-shot `collections` or `get` must not start -// downloading the whole account — and the refresh interval is long so no second -// refresh fires mid-command. -const openReadLibrary = (client: Client): Promise<Library> => - Library.open({ - client: readLibraryClient(client), - cacheDirectory: cacheDirOption(), - refreshIntervalSeconds: 3600, - precacheThumbnails: false, - precacheOriginals: false, - }); - -// Close the library and exit once stdout/stderr have drained. `process.exit` -// alone can truncate buffered piped output, and the library keeps the event -// loop alive with a background refresh, so a plain return could hang; this does -// neither. -const finish = (lib: Library | undefined, code: number): void => { - lib?.close(); - const pending = [stdout, stderr].filter((s) => s.writableLength > 0); - if (pending.length === 0) { - process.exit(code); - return; - } - let remaining = pending.length; - for (const s of pending) { - s.once("drain", () => { - if (--remaining === 0) process.exit(code); - }); - } -}; +const run = (command: Promise<number>): Promise<void> => + runCommand(command, stdout, stderr, (code) => process.exit(code)); program .command("login") .description("Log in to an Ente account and save the session") - .action(async () => { - await init(); - const email = process.env.QUAK_EMAIL ?? (await prompt("Email")); - const password = - process.env.QUAK_PASSWORD ?? (await promptSecret("Password")); - - stderr.write("Authenticating...\n"); - try { - const client = await Client.login({ - email, - password, - totp: async () => prompt("TOTP code: "), - emailOTP: async () => prompt("Email verification code: "), - }); - - saveSession(client.toJSON()); - const info = client.whoami(); - stderr.write(`Logged in as ${info.email} (user ${info.userID})\n`); - stderr.write(`Session saved to ${sessionPath}\n`); - } catch (err) { - stderr.write( - `Login failed: ${err instanceof Error ? err.message : err}\n`, - ); - process.exit(1); - } - }); + .action(() => run(loginCommand(context()))); program .command("whoami") .description("Print the logged-in account") - .action(() => { - const client = requireSession(); - const info = client.whoami(); - stdout.write(JSON.stringify(info) + "\n"); - }); + .action(() => run(whoamiCommand(context()))); program .command("logout") - .description("Delete the saved session") - .action(async () => { - if (existsSync(sessionPath)) { - const { unlinkSync } = await import("node:fs"); - unlinkSync(sessionPath); - stderr.write("Session deleted.\n"); - } else { - stderr.write("No session found.\n"); - } - }); + .description("End the session on the server and delete the saved session") + .action(() => run(logoutCommand(context()))); program .command("collections") .description("List all collections (albums)") .option("--json", "Output as JSON array") - .action(async (opts: { json?: boolean }) => { - await init(); - const client = requireSession(); - const lib = await openReadLibrary(client); - // Force a server round-trip and list in enumeration order (issue #36 - // amendment, issue #52): the pre-library CLI printed current state in - // this order, not the albums projection's newest-first order. - const collections = await freshCollections(lib); - - if (opts.json) { - stdout.write( - JSON.stringify( - collections.map((c) => ({ - id: c.id, - name: c.name, - type: c.type, - ownerID: c.ownerID, - isShared: c.isShared, - updationTime: c.updationTime, - })), - null, - 2, - ) + "\n", - ); - } else { - for (const c of collections) { - stdout.write( - `${c.id}\t${c.type}\t${c.name}${c.isShared ? " (shared)" : ""}\n`, - ); - } - } - finish(lib, 0); - }); + .action((opts: { json?: boolean }) => + run(collectionsCommand(context(), opts)), + ); program .command("files") @@ -222,39 +83,9 @@ program "Collection ID (from `quak collections`)", ) .option("--json", "Output as JSON array") - .action(async (opts: { collection: string; json?: boolean }) => { - await init(); - const client = requireSession(); - const collectionID = Number(opts.collection); - if (!Number.isFinite(collectionID)) { - stderr.write("Invalid collection ID\n"); - process.exit(1); - } - - const lib = await openReadLibrary(client); - // Force a server round-trip and list in enumeration order (issue #36 - // amendment, issue #52). Each file prints from its own decrypted - // metadata (raw title, microsecond creationTime) via cli-output, and in - // the pre-library CLI's enumeration order, not the projection's - // newest-first order. - const files = await freshFiles(lib, collectionID); - if (!files) { - stderr.write(`Collection ${collectionID} not found\n`); - finish(lib, 1); - return; - } - - if (opts.json) { - stdout.write( - JSON.stringify(files.map(fileListRow), null, 2) + "\n", - ); - } else { - for (const file of files) { - stdout.write(fileListLine(file) + "\n"); - } - } - finish(lib, 0); - }); + .action((opts: { collection: string; json?: boolean }) => + run(filesCommand(context(), opts)), + ); program .command("get") @@ -262,34 +93,9 @@ program .argument("<fileID>", "File ID (from `quak files`)") .option("--out <path>", "Output file path") .option("--collection <id>", "Accepted for compatibility; ignored") - .action(async (fileIDStr: string, opts: { out?: string }) => { - await init(); - const client = requireSession(); - const fileID = Number(fileIDStr); - if (!Number.isFinite(fileID)) { - stderr.write("Invalid file ID\n"); - process.exit(1); - } - - const lib = await openReadLibrary(client); - // Force a server round-trip so the file resolves against current state - // (issue #36 amendment, issue #52). - const resolved = await freshFile(lib, fileID); - if (!resolved) { - stderr.write(`File ${fileID} not found\n`); - finish(lib, 1); - return; - } - const { photo, file } = resolved; - - const result = await photo.original(); - // Default name is the file's own title, as the pre-library CLI used - // (not the editedName-preferring projection title) (issue #52). - const outPath = opts.out ?? originalName(file); - copyFileSync(result.path, outPath); - stderr.write(`${result.bytes} bytes -> ${outPath}\n`); - finish(lib, 0); - }); + .action((fileID: string, opts: { out?: string }) => + run(getCommand(context(), fileID, opts)), + ); program .command("get-thumb") @@ -297,34 +103,9 @@ program .argument("<fileID>", "File ID (from `quak files`)") .option("--out <path>", "Output file path") .option("--collection <id>", "Accepted for compatibility; ignored") - .action(async (fileIDStr: string, opts: { out?: string }) => { - await init(); - const client = requireSession(); - const fileID = Number(fileIDStr); - if (!Number.isFinite(fileID)) { - stderr.write("Invalid file ID\n"); - process.exit(1); - } - - const lib = await openReadLibrary(client); - // Force a server round-trip so the file resolves against current state - // (issue #36 amendment, issue #52). - const resolved = await freshFile(lib, fileID); - if (!resolved) { - stderr.write(`File ${fileID} not found\n`); - finish(lib, 1); - return; - } - const { photo, file } = resolved; - - const result = await photo.thumbnail(); - // Default name is thumb_<file's own title>, as the pre-library CLI - // used (not the projection title) (issue #52). - const outPath = opts.out ?? thumbnailName(file); - copyFileSync(result.path, outPath); - stderr.write(`${result.bytes} bytes -> ${outPath}\n`); - finish(lib, 0); - }); + .action((fileID: string, opts: { out?: string }) => + run(getThumbCommand(context(), fileID, opts)), + ); program .command("backup-metadata") @@ -337,16 +118,9 @@ program "Download each file and extract full EXIF/IPTC/XMP metadata (slow)", ) .option("--all", "Alias for --exif") - .action(async (dir: string, opts: { exif?: boolean; all?: boolean }) => { - await init(); - const client = requireSession(); - const lib = await openReadLibrary(client); - await runMetadataBackup(lib, client, dir, { - exif: opts.exif || opts.all, - onProgress: (msg) => stderr.write(msg + "\n"), - }); - finish(lib, 0); - }); + .action((dir: string, opts: { exif?: boolean; all?: boolean }) => + run(backupMetadataCommand(context(), dir, opts)), + ); program .command("backup") @@ -355,43 +129,9 @@ program ) .argument("<dir>", "Output directory") .option("--json", "Print result as JSON instead of human-readable summary") - .action(async (dir: string, opts: { json?: boolean }) => { - await init(); - const client = requireSession(); - - stderr.write("Starting backup...\n"); - const lib = await Library.open({ - client, - downloadDirectory: dir, - cacheDirectory: cacheDirOption(), - }); - const result = await lib.backup({ - downloadDirectory: dir, - onProgress: (msg) => { - if (!opts.json) stderr.write(msg + "\n"); - }, - }); - - if (opts.json) { - stdout.write(JSON.stringify(result, null, 2) + "\n"); - } else { - stderr.write("\n--- Backup complete ---\n"); - stderr.write(` Total files: ${result.totalFiles}\n`); - stderr.write(` Downloaded: ${result.downloaded}\n`); - stderr.write(` Skipped: ${result.skipped}\n`); - stderr.write(` Failed: ${result.failed}\n`); - if (result.errors.length > 0) { - stderr.write("\nFailed files:\n"); - for (const e of result.errors) { - stderr.write( - ` [${e.collection}] ${e.title} (id ${e.fileID}): ${e.error}\n`, - ); - } - } - } - - finish(lib, result.failed > 0 ? 1 : 0); - }); + .action((dir: string, opts: { json?: boolean }) => + run(backupCommand(context(), dir, opts)), + ); const helper = program .command("helper") @@ -401,32 +141,9 @@ helper .command("list-missing-thumbnails") .description("List files whose thumbnails are missing or empty") .option("--json", "Output as JSON array") - .action(async (opts: { json?: boolean }) => { - await init(); - const client = requireSession(); - const lib = await openReadLibrary(client); - const missing = await listMissingThumbnails(lib, client, (msg) => { - if (!opts.json) stderr.write(msg + "\n"); - }); - - if (opts.json) { - stdout.write(JSON.stringify(missing, null, 2) + "\n"); - } else { - if (missing.length === 0) { - stderr.write("No missing thumbnails found.\n"); - } else { - stderr.write( - `\n${missing.length} file(s) with missing thumbnails:\n`, - ); - for (const m of missing) { - stdout.write( - `${m.fileID}\t${m.title}\t${m.collection}\t${m.reason}\n`, - ); - } - } - } - finish(lib, 0); - }); + .action((opts: { json?: boolean }) => + run(listMissingThumbnailsCommand(context(), opts)), + ); helper .command("fix-missing-thumbnails") @@ -438,65 +155,9 @@ helper "Specific file IDs to fix (default: fix all missing)", ) .option("--json", "Output as JSON") - .action(async (opts: { file?: string[]; json?: boolean }) => { - await init(); - const client = requireSession(); - const lib = await openReadLibrary(client); - - let fileIDs: number[]; - if (opts.file && opts.file.length > 0) { - fileIDs = opts.file.map(Number).filter(Number.isFinite); - } else { - stderr.write("Scanning for missing thumbnails...\n"); - const missing = await listMissingThumbnails(lib, client, (msg) => { - if (!opts.json) stderr.write(msg + "\n"); - }); - fileIDs = missing.map((m) => m.fileID); - if (fileIDs.length === 0) { - stderr.write("No missing thumbnails found.\n"); - finish(lib, 0); - return; - } - stderr.write(`Found ${fileIDs.length} file(s) to fix.\n`); - } - - const results = await fixMissingThumbnails( - lib, - client, - fileIDs, - (msg) => { - if (!opts.json) stderr.write(msg + "\n"); - }, - ); - - if (opts.json) { - stdout.write(JSON.stringify(results, null, 2) + "\n"); - } else { - const fixed = results.filter((r) => r.status === "fixed").length; - const skipped = results.filter( - (r) => r.status === "skipped", - ).length; - const failed = results.filter((r) => r.status === "failed").length; - stderr.write(`\n--- Done ---\n`); - stderr.write(` Fixed: ${fixed}\n`); - stderr.write(` Skipped: ${skipped}\n`); - stderr.write(` Failed: ${failed}\n`); - if (skipped > 0) { - stderr.write("\nSkipped (unsupported format):\n"); - for (const r of results.filter((r) => r.status === "skipped")) { - stderr.write(` ${r.fileID}\t${r.title}\t${r.reason}\n`); - } - } - if (failed > 0) { - stderr.write("\nFailed files:\n"); - for (const r of results.filter((r) => r.status === "failed")) { - stderr.write(` ${r.fileID}\t${r.title}\t${r.reason}\n`); - } - } - } - - finish(lib, results.some((r) => r.status === "failed") ? 1 : 0); - }); + .action((opts: { file?: string[]; json?: boolean }) => + run(fixMissingThumbnailsCommand(context(), opts)), + ); await init(); -program.parse(); +await program.parseAsync(); diff --git a/package.json b/package.json index 423fad4..f1a1d6f 100644 --- a/package.json +++ b/package.json @@ -9,17 +9,23 @@ "type": "git", "url": "https://git.eeqj.de/sneak/quak.git" }, + "private": true, "type": "module", + "engines": { + "node": ">=22" + }, "main": "./dist/src/index.js", "types": "./dist/src/index.d.ts", + "exports": { + ".": { + "types": "./dist/src/index.d.ts", + "import": "./dist/src/index.js" + }, + "./package.json": "./package.json" + }, "bin": { "quak": "./dist/bin/quak.js" }, - "files": [ - "dist/", - "README.md", - "LICENSE" - ], "scripts": { "build": "script/build", "quak": "node ./dist/bin/quak.js", @@ -29,7 +35,6 @@ }, "devDependencies": { "@eslint/js": "9.38.0", - "@types/libsodium-wrappers-sumo": "0.8.2", "@types/node": "22.18.13", "eslint": "9.38.0", "prettier": "3.8.1", @@ -43,6 +48,7 @@ "env-paths": "4.0.0", "exif-reader": "2.0.3", "fast-srp-hap": "2.0.4", + "fflate": "0.8.3", "jpeg-js": "0.4.4", "libsodium-wrappers-sumo": "0.8.4" } diff --git a/script/build b/script/build index 05d5180..0a074b7 100755 --- a/script/build +++ b/script/build @@ -45,10 +45,24 @@ for (const bin of bins) { ' } +# src/index.ts imports ../package.json for the version, which tsc copies to +# dist/package.json. Running the built CLI proves that import resolves from +# dist/ and reports the version package.json declares. +verify_version() { + built="$(node dist/bin/quak.js --version)" + declared="$(node -p 'require("./package.json").version')" + if [ "$built" != "$declared" ]; then + echo "build: dist/bin/quak.js reports $built, package.json declares $declared" >&2 + exit 1 + fi + echo "build: dist/bin/quak.js reports version $built" +} + main() { cd "$ROOT" yarn run tsc verify_entrypoints + verify_version } main "$@" diff --git a/script/check b/script/check index 2f37914..1607369 100755 --- a/script/check +++ b/script/check @@ -1,19 +1,11 @@ #!/bin/sh # script/check: run all checks (test, lint). Our own extension to -# scripts-to-rule-them-all. Must not modify any files. +# scripts-to-rule-them-all. Both are Docker phases. Must not modify any +# files. # -# The formatting check is part of lint, not a step of its own: -# script/lint builds Dockerfile.lint, which runs eslint AND -# `prettier --check .` as build steps. Calling script/fmt-check here as -# well would run prettier a second time over the same tree for the same -# verdict — the weaker of the two, since the host toolchain is whatever -# the working tree happens to have installed while the container's is -# digest-pinned. script/fmt-check remains a standalone entrypoint for -# asking the formatting question by itself. -# -# script/lint builds Dockerfile.lint, so this script requires docker and -# must never be run from inside a container: that is why the Dockerfile -# image runs script/test and script/build rather than this. +# script/fmt-check is not called here, unlike the template: the lint +# phase already runs `prettier --check .`, so calling it would run +# prettier a second time over the same tree for the same verdict. set -eu SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)" diff --git a/script/cibuild b/script/cibuild index dd1f7b5..38f5706 100755 --- a/script/cibuild +++ b/script/cibuild @@ -1,15 +1,11 @@ #!/bin/sh -# script/cibuild: run the CI build, which is both images in a defined order. -# -# First script/lint, which builds Dockerfile.lint and is the one and only -# place linting happens — it goes first so a lint failure is reported before -# the slower suite runs. Then the Dockerfile image, which runs script/test -# and script/build. CHECK_EPOCH and LINT_EPOCH differ on every invocation, so -# neither the linters nor the suite can be served from Docker's cache: a -# green build here means the checks ran now, not that a previous run was -# remembered. The layers below the epochs (bootstrap, yarn install) are -# unaffected and stay cached. A build that omits the arguments fails by -# design. +# script/cibuild: run the CI build. The image's last stage depends on the +# lint and test phases, so this one build runs eslint, prettier and the +# suite once each and then compiles. Unlike the template it does not run +# script/check first, which would run lint and the tests a second time. +# --no-cache for the same reason as script/docker: the gate phases the +# final stage depends on are RUN steps, and a cached one is a check that +# did not run. set -eu SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)" @@ -17,8 +13,16 @@ ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)" main() { cd "$ROOT" - "$SCRIPT_DIR/lint" - docker build --build-arg CHECK_EPOCH="$(date +%s)" . + # Own line: a failing command substitution inside an argument does + # not trip `set -e`, so the inline form degrades silently to an + # empty constant. VERSION is computed here because .dockerignore + # excludes .git, so `git describe` in a build stage yields an empty + # version without failing. + version="$(git describe --tags --always --dirty 2>/dev/null || true)" + [ -n "$version" ] || version="unknown" + docker build --no-cache \ + --build-arg VERSION="$version" \ + -t "$("$SCRIPT_DIR/projectname")" . } main "$@" diff --git a/script/docker b/script/docker index c691f2c..c4688e8 100755 --- a/script/docker +++ b/script/docker @@ -1,10 +1,8 @@ #!/bin/sh # script/docker: build the Docker image tagged with the project name. # Identical in all repos; the tag comes from script/projectname. -# CHECK_EPOCH is passed for the same reason script/cibuild passes it: the -# Dockerfile refuses to build without it, so that no path to an image can -# quietly serve the test and build layers from cache. This builds the test -# and build image only; linting is a separate image, built by script/lint. +# --no-cache because the gate phases the final stage depends on are RUN +# steps, and a cached one is a check that did not run. set -eu SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)" @@ -12,7 +10,15 @@ ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)" main() { cd "$ROOT" - docker build --build-arg CHECK_EPOCH="$(date +%s)" \ + # Own line: a failing command substitution inside an argument does + # not trip `set -e`, so the inline form degrades silently to an + # empty constant. VERSION is computed here because .dockerignore + # excludes .git, so `git describe` in a build stage yields an empty + # version without failing. + version="$(git describe --tags --always --dirty 2>/dev/null || true)" + [ -n "$version" ] || version="unknown" + docker build --no-cache \ + --build-arg VERSION="$version" \ -t "$("$SCRIPT_DIR/projectname")" . } diff --git a/script/lint b/script/lint index 4a602f2..2d8b075 100755 --- a/script/lint +++ b/script/lint @@ -1,24 +1,23 @@ #!/bin/sh -# script/lint: run the linters. eslint and prettier are never run against -# the working tree from here: linting runs via docker only, one way, -# everywhere — script/lint builds Dockerfile.lint, which COPYs the repo into -# the pinned node image and runs the linters as build steps. That works even -# when the docker daemon is remote and bind mounts are impossible. +# script/lint: run the linter. Linting is a phase of the Dockerfile and +# this builds that phase alone; the linter is never installed or run on +# a developer host, where a shared result cache and a host-global lock +# make its answer untrustworthy. # -# LINT_EPOCH is passed on every invocation because no lint cache is wanted: -# on an unchanged tree Docker would otherwise serve the linter layers, having -# linted nothing, and still exit 0. Dockerfile.lint refuses to build without -# the argument, so no path to a lint result can quietly come from cache. -# -# Nothing that runs inside a container may call this script; see the header -# of Dockerfile. +# The phase is not the last stage in the file, so it is built only when +# --target names it. --no-cache because a cached lint layer is a lint +# that did not run. The tag makes each build replace the previous image +# instead of leaving a dangling one behind. set -eu -ROOT="$(cd "$(dirname "$0")/.." && pwd -P)" +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)" +ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)" main() { cd "$ROOT" - docker build --build-arg LINT_EPOCH="$(date +%s)" -f Dockerfile.lint . + docker build --no-cache \ + --target lint \ + -t "$("$SCRIPT_DIR/projectname")-lint" . } main "$@" diff --git a/script/precommit b/script/precommit index 489e72d..e87c5c8 100755 --- a/script/precommit +++ b/script/precommit @@ -4,17 +4,9 @@ # # Runs lint but deliberately NOT the tests, so the TDD red-phase commit # (failing tests, no implementation yet) can land. CI runs -# script/cibuild, which builds both images and so catches any branch -# that ships red. -# -# The formatting check is still enforced here, because script/lint is a -# build of Dockerfile.lint and that runs `prettier --check .` as a build -# step: a badly formatted tree fails this hook, and therefore the -# commit. Calling script/fmt-check as well would only run prettier a -# second time over the same tree for the same verdict. -# -# script/lint is a docker build (Dockerfile.lint); docker is required to -# commit, which is the point of linting one way, everywhere. +# script/cibuild, whose image build includes the test phase, and so +# catches any branch that ships red. The lint phase includes the +# prettier check, so a badly formatted tree still fails the commit. set -eu SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)" diff --git a/script/test b/script/test index d4cec4b..cd239f2 100755 --- a/script/test +++ b/script/test @@ -1,25 +1,19 @@ #!/bin/sh -# script/test: run the test suite. Uses `timeout` (GNU coreutils) when -# available so the run is hard-capped at 30s; on macOS without -# coreutils the cap is skipped. +# script/test: run the test suite. Testing is a phase of the Dockerfile +# and this builds that phase alone, on the same terms as script/lint: +# --target because a phase that is not the last stage is built only when +# named, --no-cache because a cached test layer is a test that did not +# run, and a tag so each build replaces the previous image. set -eu -ROOT="$(cd "$(dirname "$0")/.." && pwd -P)" - -rerun_verbose() { - echo "--- Rerunning with verbose for details ---" - yarn run vitest run --reporter=verbose - exit 1 -} +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd -P)" +ROOT="$(cd "$SCRIPT_DIR/.." && pwd -P)" main() { cd "$ROOT" - TIMEOUT="$(command -v timeout 2>/dev/null || command -v gtimeout 2>/dev/null || true)" - if [ -n "$TIMEOUT" ]; then - "$TIMEOUT" 30s yarn run vitest run --reporter=dot || rerun_verbose - else - yarn run vitest run --reporter=dot || rerun_verbose - fi + docker build --no-cache \ + --target test \ + -t "$("$SCRIPT_DIR/projectname")-test" . } main "$@" diff --git a/src/api/client.ts b/src/api/client.ts index f3b9996..316f9cc 100644 --- a/src/api/client.ts +++ b/src/api/client.ts @@ -19,14 +19,12 @@ const DEFAULT_FILES_ORIGIN = "https://files.ente.io"; const DEFAULT_THUMBS_ORIGIN = "https://thumbnails.ente.io"; const CLIENT_PACKAGE = "berlin.sneak.quak"; -// Two deadlines rather than one, because a single number cannot serve both -// jobs. Thirty seconds is generous for a JSON call and short enough that a -// hung API connection cannot stall a backup for long. A file body is a -// different shape of problem: the deadline has to cover the whole transfer, -// which for a large video on a slow link is minutes, so a value sane for JSON -// would cancel legitimate downloads. +// Two deadlines of different kinds. `requestTimeoutMs` bounds a whole JSON +// call. `downloadTimeoutMs` is an idle deadline: a file or thumbnail download +// is aborted only when no bytes have arrived for that long, so a large video on +// a slow link that keeps making progress is never cut off. export const DEFAULT_REQUEST_TIMEOUT_MS = 30_000; -export const DEFAULT_DOWNLOAD_TIMEOUT_MS = 600_000; +export const DEFAULT_DOWNLOAD_TIMEOUT_MS = 60_000; export interface ApiClientOptions { apiOrigin?: string; @@ -48,18 +46,44 @@ export interface StreamOptions { retry?: boolean; } -// Enforce a deadline over a response body, not merely over its headers. +// An abort signal that fires once `ms` pass without a call to `restart`. It +// aborts with a `TimeoutError`, the same reason `AbortSignal.timeout()` gives, +// so the retry classifier treats an idle download exactly as it treats any +// other deadline. `stop` must be called when the download ends. The timer is +// unref'd, so even one left running never keeps the process alive. +const idleDeadline = (ms: number) => { + const controller = new AbortController(); + let timer: ReturnType<typeof setTimeout> | undefined; + const stop = (): void => clearTimeout(timer); + const restart = (): void => { + stop(); + timer = setTimeout(() => { + controller.abort( + new DOMException( + `download stalled: no bytes received for ${ms} ms`, + "TimeoutError", + ), + ); + }, ms); + timer.unref(); + }; + restart(); + return { signal: controller.signal, restart, stop }; +}; + +// Enforce the idle deadline over a response body, not merely over its headers. // // `getFileStream` returns as soon as headers arrive; the bytes are pulled // later, in the download layer. Whether the signal passed to `fetch` also // tears down the body afterwards is up to the fetch implementation, so this // wrapper makes it a property of quak instead: every read races the signal, -// and an abort errors the stream with the abort reason — which the retry -// classifier recognises. +// each chunk that arrives restarts the deadline, and an abort errors the +// stream with the abort reason — which the retry classifier recognises. const deadlineStream = ( body: ReadableStream<Uint8Array>, - signal: AbortSignal, + deadline: ReturnType<typeof idleDeadline>, ): ReadableStream<Uint8Array> => { + const { signal } = deadline; const reader = body.getReader(); let rejectOnAbort: (reason: unknown) => void = () => undefined; const aborted = new Promise<never>((_resolve, reject) => { @@ -73,7 +97,10 @@ const deadlineStream = ( const onAbort = (): void => rejectOnAbort(signal.reason); if (signal.aborted) onAbort(); else signal.addEventListener("abort", onAbort, { once: true }); - const release = (): void => signal.removeEventListener("abort", onAbort); + const release = (): void => { + deadline.stop(); + signal.removeEventListener("abort", onAbort); + }; return new ReadableStream<Uint8Array>({ async pull(controller) { @@ -84,6 +111,7 @@ const deadlineStream = ( controller.close(); return; } + deadline.restart(); controller.enqueue(next.value); } catch (err) { release(); @@ -98,6 +126,30 @@ const deadlineStream = ( }); }; +// The one place a request URL is built. `origin` may carry a base path (a +// self-hosted server behind a prefix) and may end in a slash; `path` may or +// may not start with one. Query parameters go only through `query`, which +// percent-encodes them: a `?` or `#` in `path` is an error, because +// `new URL` would otherwise quietly treat what follows as something else. +const buildURL = ( + origin: string, + path: string, + query?: Record<string, string | number | undefined>, +): string => { + if (path.includes("?") || path.includes("#")) { + throw new Error( + `request path must not contain "?" or "#"; pass query parameters separately: ${path}`, + ); + } + const url = new URL( + `${origin.replace(/\/+$/, "")}/${path.replace(/^\/+/, "")}`, + ); + for (const [k, v] of Object.entries(query ?? {})) { + if (v !== undefined) url.searchParams.set(k, String(v)); + } + return url.href; +}; + export class ApiClient { private readonly apiOrigin: string; private readonly isCustomOrigin: boolean; @@ -139,11 +191,16 @@ export class ApiClient { this.token = undefined; } + getAuthToken(): string | undefined { + return this.token; + } + // The policy this client was configured with, so that a caller wrapping a // whole operation in its own `withRetry` — the download layer — runs under // the same settings rather than under the library defaults. + // A copy, so the caller cannot change this client's settings through it. getRetryOptions(): ResolvedRetryOptions { - return this.retry; + return { ...this.retry }; } private headers(extra?: Record<string, string>): Record<string, string> { @@ -197,22 +254,10 @@ export class ApiClient { path: string, query?: Record<string, string | number | undefined>, ): Promise<T> { - const url = new URL(path, this.apiOrigin + "/"); - // new URL with a base resolves relative paths; ensure we keep the - // origin from apiOrigin even when path starts with / - url.protocol = new URL(this.apiOrigin).protocol; - url.host = new URL(this.apiOrigin).host; - url.pathname = path; - if (query) { - for (const [k, v] of Object.entries(query)) { - if (v !== undefined) { - url.searchParams.set(k, String(v)); - } - } - } + const url = buildURL(this.apiOrigin, path, query); // A GET changes nothing, so it is retried under the full policy. return withRetry(async () => { - const resp = await this._fetch(url.href, { + const resp = await this._fetch(url, { method: "GET", headers: this.headers(), signal: AbortSignal.timeout(this.requestTimeoutMs), @@ -223,16 +268,16 @@ export class ApiClient { } async postJSON<T>(path: string, body: unknown): Promise<T> { - const url = `${this.apiOrigin}${path}`; - // Idempotency: this reaches `/users/srp/create-session`, - // `/users/two-factor/verify` and `/users/ott`, all of which change - // server state — verifying a second factor consumes one of a small - // number of attempts. So a POST is replayed only on a failure that - // establishes no TCP connection to the server ever existed: DNS - // produced no address, or the peer refused the connection. A 5xx, a - // mid-flight reset, a routing errno (which Linux also delivers on an - // established socket) and a timeout are all left to the caller, - // because each of them can occur after the server has already acted. + const url = buildURL(this.apiOrigin, path); + // Not idempotent: a POST is replayed only when `isSafeToReplay` + // says no request byte can have reached the server. The endpoints + // this covers are listed in the README under "Endpoints used". + // + // Redirects are not followed. The origin has already received the + // request when it answers with one, so a connection refused by the + // redirect target would look replay-safe when it is not. The API has + // no legitimate redirect, so one surfaces as an `ApiError` with its + // 3xx status, which is not retried. return withRetry( async () => { const resp = await this._fetch(url, { @@ -241,6 +286,7 @@ export class ApiClient { "Content-Type": "application/json", }), body: JSON.stringify(body), + redirect: "manual", signal: AbortSignal.timeout(this.requestTimeoutMs), }); await this.throwIfError(resp); @@ -255,8 +301,8 @@ export class ApiClient { opts?: StreamOptions, ): Promise<ReadableStream<Uint8Array>> { const url = this.isCustomOrigin - ? `${this.apiOrigin}/files/download/${fileID}` - : `${this.filesOrigin}/?fileID=${fileID}`; + ? buildURL(this.apiOrigin, `/files/download/${fileID}`) + : buildURL(this.filesOrigin, "/", { fileID }); return this.streamRequest(url, opts); } @@ -298,10 +344,8 @@ export class ApiClient { } async putJSON<T>(path: string, body: unknown): Promise<T> { - const url = `${this.apiOrigin}${path}`; - // Same idempotency rule as `postJSON`, for the same reason: this - // reaches `/files/thumbnail`, which registers an uploaded thumbnail - // against a file. + const url = buildURL(this.apiOrigin, path); + // Same replay and redirect rules as `postJSON`, for the same reasons. return withRetry( async () => { const resp = await this._fetch(url, { @@ -310,6 +354,7 @@ export class ApiClient { "Content-Type": "application/json", }), body: JSON.stringify(body), + redirect: "manual", signal: AbortSignal.timeout(this.requestTimeoutMs), }); await this.throwIfError(resp); @@ -335,8 +380,8 @@ export class ApiClient { opts?: StreamOptions, ): Promise<ReadableStream<Uint8Array>> { const url = this.isCustomOrigin - ? `${this.apiOrigin}/files/preview/${fileID}` - : `${this.thumbsOrigin}/?fileID=${fileID}`; + ? buildURL(this.apiOrigin, `/files/preview/${fileID}`) + : buildURL(this.thumbsOrigin, "/", { fileID }); return this.streamRequest(url, opts); } @@ -345,22 +390,27 @@ export class ApiClient { opts?: StreamOptions, ): Promise<ReadableStream<Uint8Array>> { const once = async (): Promise<ReadableStream<Uint8Array>> => { - // A fresh deadline per attempt, so a retry gets the whole budget - // rather than the remainder of the one that just expired. - const signal = AbortSignal.timeout(this.downloadTimeoutMs); - const resp = await this._fetch(url, { - method: "GET", - headers: this.headers(), - signal, - }); - await this.throwIfError(resp); - if (!resp.body) { - // Carries the status, and is not retryable: a response that - // arrived without a body is malformed, and asking again - // produces the same malformed response. - throw new ApiError("response body is null", resp.status); + // A fresh deadline per attempt. It also covers the wait for the + // headers, when no bytes have arrived either. + const deadline = idleDeadline(this.downloadTimeoutMs); + try { + const resp = await this._fetch(url, { + method: "GET", + headers: this.headers(), + signal: deadline.signal, + }); + await this.throwIfError(resp); + if (!resp.body) { + // Carries the status, and is not retryable: a response + // that arrived without a body is malformed, and asking + // again produces the same malformed response. + throw new ApiError("response body is null", resp.status); + } + return deadlineStream(resp.body, deadline); + } catch (err) { + deadline.stop(); + throw err; } - return deadlineStream(resp.body, signal); }; return opts?.retry === false ? once() : withRetry(once, this.retry); } diff --git a/src/backup.ts b/src/backup.ts index 7d26ef8..019619c 100644 --- a/src/backup.ts +++ b/src/backup.ts @@ -1,9 +1,11 @@ // The backup command, rebuilt on the library API (issue #51). // -// `lib.backup()` refreshes the library, then, for every file in scope, gets its -// original bytes onto disk under `downloadDirectory` and rebuilds the derived -// views (per-file sidecars, per-collection symlink trees, per-collection JSON) -// from the model. The on-disk layout is the historical one, unchanged: +// `lib.backup()` waits for a completed refresh of the library (a failed one +// fails the backup before any file is touched), then, for every file in scope, +// gets its original bytes onto disk under `downloadDirectory` and rebuilds the +// derived views (per-file sidecars, per-collection symlink trees, +// per-collection JSON) from the model. The on-disk layout is the historical +// one, unchanged: // // <downloadDirectory>/ // originals/<fileID>.<ext> the decrypted bytes @@ -17,7 +19,9 @@ // temp-then-rename, so a file that exists is whole and is never re-fetched — an // interrupted run resumes by listing the directory. The derived views hold no // unique state, so they are rebuilt every run; that repairs stale sidecars and -// missing or broken symlinks left by an earlier crash. +// missing or broken symlinks left by an earlier crash. A rebuild also removes +// the symlinks into originals/ that no longer belong to an album, and the +// directories of albums that no longer exist. // // Resilience (issue #8): no per-file condition aborts the run. A failed // download or a failed symlink is caught, recorded in `failures.json` with a @@ -29,19 +33,22 @@ // rather than counted forever, which would poison a scheduled backup's exit code. import { - copyFileSync, lstatSync, mkdirSync, + readdirSync, readFileSync, readlinkSync, - renameSync, + rmdirSync, rmSync, statSync, symlinkSync, writeFileSync, } from "node:fs"; +import { copyFile, rename, rm } from "node:fs/promises"; import { basename, dirname, extname, join, relative } from "node:path"; +import { fsyncPath, removeLeftoverTempFiles } from "./download/index.js"; +import { safeExtension, sanitizeFileName } from "./filename.js"; import type { Collection, EnteFile } from "./model/types.js"; export type ProgressCallback = (message: string) => void; @@ -90,8 +97,9 @@ export interface BackupLibrary { listCollections(): Collection[]; listFiles(collectionID: number): EnteFile[]; // Get an original's bytes onto disk through the content cache/pools, - // returning where they landed (the cache, or a prior backup). - original(fileID: number): Promise<{ path: string }>; + // returning where they landed: `destination` when they were fetched now, + // otherwise wherever they already were (the cache, or a prior backup). + original(fileID: number, destination: string): Promise<{ path: string }>; thumbnail(fileID: number): Promise<{ path: string }>; } @@ -108,16 +116,11 @@ interface FailureEntry { const LEDGER_VERSION = 1; -const sanitizePath = (name: string): string => - name.replace(/[/\\:*?"<>|]/g, "_").replace(/^\.+/, "_"); - // The originals/ filename for a file: `<id><ext>`, the extension taken from the // title (or `.bin`). Matches the content cache's own naming so a present check // lines up with what a fetch would write. -const originalName = (file: EnteFile): string => { - const ext = extname(file.metadata.title || "") || ".bin"; - return `${file.id}${ext}`; -}; +const originalName = (file: EnteFile): string => + `${file.id}${safeExtension(file.metadata.title)}`; // A regular file with content is treated as complete. A zero-byte file is not: // it is the shape an aborted write leaves and must be re-fetched. @@ -158,8 +161,12 @@ const errorMessage = (err: unknown): string => err instanceof Error ? err.message : String(err); // Copy bytes into `dest` via a temp file in the same directory plus rename, so -// `dest` appears only once it is whole ("present means complete"). -const copyAtomic = (src: string, dest: string): void => { +// `dest` appears only once it is whole ("present means complete"). As in the +// download writer, the temp file is fsynced before the rename and the directory +// after it, so a power cut cannot leave a correctly named but short original. +// The temp name carries this process's ID so a later run can tell a leftover +// from a copy still in progress (see `removeLeftoverTempFiles`). +const copyAtomic = async (src: string, dest: string): Promise<void> => { if (src === dest) return; const tmp = join( dirname(dest), @@ -168,10 +175,15 @@ const copyAtomic = (src: string, dest: string): void => { .slice(2)}.tmp`, ); try { - copyFileSync(src, tmp); - renameSync(tmp, dest); + await copyFile(src, tmp); + await fsyncPath(tmp); + // `rename` replaces the destination's directory entry: an existing + // symlink at `dest` is replaced, not followed, and the new file has + // the temp file's permissions (copied from `src`). + await rename(tmp, dest); + await fsyncPath(dirname(dest)); } finally { - rmSync(tmp, { force: true }); + await rm(tmp, { force: true }); } }; @@ -191,6 +203,93 @@ const rebuildSymlink = (linkPath: string, target: string): void => { symlinkSync(target, linkPath); }; +// The on-disk names for the entries of one directory, keyed by ID. Each name +// is used as is unless another entry would get the same name, ignoring case +// (two names that differ only in case are one entry on a case-insensitive +// file system); then every entry sharing it gets ` (<id>)`, before the +// extension when `beforeExtension` is set. A name with an ID added can match +// another entry's own name (`IMG (6).JPG`), so this repeats until no name is +// shared. IDs are stable, so the names are too. +const namesByID = ( + entries: { id: number; name: string }[], + beforeExtension: boolean, +): Map<number, string> => { + const withID = (id: number, name: string): string => { + const ext = beforeExtension ? extname(name) : ""; + const stem = name.slice(0, name.length - ext.length); + return `${stem} (${id})${ext}`; + }; + const names = new Map<number, string>(); + for (const { id, name } of entries) names.set(id, name); + const suffixed = new Set<number>(); + for (;;) { + const counts = new Map<string, number>(); + for (const name of names.values()) { + const key = name.toLowerCase(); + counts.set(key, (counts.get(key) ?? 0) + 1); + } + let changed = false; + for (const { id, name } of entries) { + if (suffixed.has(id)) continue; + if (counts.get(name.toLowerCase()) === 1) continue; + names.set(id, withID(id, name)); + suffixed.add(id); + changed = true; + } + if (!changed) return names; + } +}; + +// Remove the symlinks in the album directory `dir` that point into +// `originalsDir` and are not named in `keep`. Nothing else in the directory +// is touched: anything else there was put there by the user. +const removeStaleLinks = ( + dir: string, + keep: Set<string>, + originalsDir: string, +): void => { + const target = relative(dir, originalsDir); + for (const name of readdirSync(dir)) { + if (keep.has(name)) continue; + const path = join(dir, name); + if ( + lstatSync(path).isSymbolicLink() && + dirname(readlinkSync(path)) === target + ) { + rmSync(path); + } + } +}; + +// Remove the directories under `collectionsDir` that an earlier run wrote for +// an album that is gone or renamed: a directory not named in `current` with a +// `<name>.json` beside it holding an album ID, which is what a run writes. Its +// symlinks into originals/ are removed; if that leaves it empty, it and its +// JSON are deleted, otherwise both stay for what the user put there. +const removeStaleAlbumDirs = ( + collectionsDir: string, + current: Set<string>, + originalsDir: string, +): void => { + for (const entry of readdirSync(collectionsDir, { withFileTypes: true })) { + if (!entry.isDirectory() || current.has(entry.name)) continue; + const jsonPath = join(collectionsDir, `${entry.name}.json`); + try { + const album = JSON.parse(readFileSync(jsonPath, "utf-8")) as { + id?: unknown; + }; + if (typeof album.id !== "number") continue; + } catch { + continue; + } + const dir = join(collectionsDir, entry.name); + removeStaleLinks(dir, new Set(), originalsDir); + if (readdirSync(dir).length > 0) continue; + rmdirSync(dir); + rmSync(jsonPath); + } +}; + const loadLedger = (path: string): Map<number, FailureEntry> => { const ledger = new Map<number, FailureEntry>(); try { @@ -258,6 +357,8 @@ export const runBackup = async ( mkdirSync(originalsDir, { recursive: true }); mkdirSync(collectionsDir, { recursive: true }); if (includeThumbnails) mkdirSync(thumbnailsDir, { recursive: true }); + removeLeftoverTempFiles(originalsDir); + removeLeftoverTempFiles(thumbnailsDir); const ledgerPath = join(downloadDirectory, "failures.json"); const ledger = loadLedger(ledgerPath); @@ -265,9 +366,10 @@ export const runBackup = async ( // Collections in scope, and the distinct files across them (a file shared // by two albums is one original). - const collections = lib - .listCollections() - .filter((c) => (only ? only.has(c.name) : true)); + const allCollections = lib.listCollections(); + const collections = allCollections.filter((c) => + only ? only.has(c.name) : true, + ); const collectionName = new Map<number, string>(); for (const c of collections) collectionName.set(c.id, c.name); @@ -324,8 +426,10 @@ export const runBackup = async ( } try { log(`Fetching original ${file.metadata.title} (${fileID})...`); - const { path } = await lib.original(fileID); - copyAtomic(path, dest); + // A fetched original is written straight to `dest`; only one + // that was already cached elsewhere is copied. + const { path } = await lib.original(fileID, dest); + await copyAtomic(path, dest); downloaded++; } catch (err) { log( @@ -346,7 +450,7 @@ export const runBackup = async ( if (isPresent(dest)) continue; try { const { path } = await lib.thumbnail(fileID); - copyAtomic(path, dest); + await copyAtomic(path, dest); } catch (err) { recordFailure( file, @@ -368,22 +472,55 @@ export const runBackup = async ( } } - // Then the per-collection symlink trees and JSON. + // Then the per-collection symlink trees and JSON. Directory names are + // chosen across every album, not just those in scope, so a scoped run + // names an album the same as a full one and never takes the directory of + // an album it skipped. Stale entries are removed before anything is + // rebuilt, so on a case-insensitive file system removing an old name can + // never remove the new one. + const albumDirNames = namesByID( + allCollections.map((c) => ({ + id: c.id, + name: sanitizeFileName(c.name, `collection-${c.id}`), + })), + false, + ); + try { + removeStaleAlbumDirs( + collectionsDir, + new Set(albumDirNames.values()), + originalsDir, + ); + } catch (err) { + log(`FAILED removing old album directories: ${errorMessage(err)}`); + } + for (const c of collections) { - const colDirName = sanitizePath(c.name || `collection-${c.id}`); + const colDirName = albumDirNames.get(c.id)!; const colDir = join(collectionsDir, colDirName); mkdirSync(colDir, { recursive: true }); const files = filesByCollection.get(c.id) ?? []; + const linkNames = namesByID( + files.map((f) => ({ + id: f.id, + name: sanitizeFileName(f.metadata.title, `file-${f.id}`), + })), + true, + ); + try { + removeStaleLinks(colDir, new Set(linkNames.values()), originalsDir); + } catch (err) { + log(`FAILED removing old links in ${c.name}: ${errorMessage(err)}`); + } + const metaFiles: { id: number; metadata: EnteFile["metadata"] }[] = []; for (const file of files) { metaFiles.push({ id: file.id, metadata: file.metadata }); if (!includeOriginals) continue; const orig = join(originalsDir, originalName(file)); if (!isPresent(orig)) continue; - const linkName = sanitizePath( - file.metadata.title || `file-${file.id}`, - ); + const linkName = linkNames.get(file.id)!; const linkPath = join(colDir, linkName); try { rebuildSymlink(linkPath, relative(colDir, orig)); diff --git a/src/cli-commands.ts b/src/cli-commands.ts new file mode 100644 index 0000000..ea4ff18 --- /dev/null +++ b/src/cli-commands.ts @@ -0,0 +1,536 @@ +// The CLI's commands as plain functions. +// +// Each command takes its options and a `CliContext` and resolves to the exit +// code; a thrown error is left to the caller. Nothing here calls +// `process.exit`: `bin/quak.ts` wires these to the command line, and `run` in +// `cli-run.ts` prints a thrown error as one line and exits once output has +// drained. Output must stay byte-identical (see `cli-output.ts`). + +import { + copyFileSync, + existsSync, + mkdirSync, + unlinkSync, + writeFileSync, +} from "node:fs"; +import { join } from "node:path"; +import { + type Client, + type ClientSnapshot, + type LoginOptions, +} from "./client.js"; +import { init } from "./crypto/index.js"; +import { + defaultCacheDirectory, + Library, + type LibraryClient, +} from "./library/index.js"; +import { + fileListRow, + fileListLine, + originalName, + thumbnailName, +} from "./cli-output.js"; +import { freshCollections, freshFiles, freshFile } from "./cli-read.js"; +import { runMetadataBackup } from "./metadata-backup.js"; +import { listMissingThumbnails, fixMissingThumbnails } from "./thumbnails.js"; + +export interface CliContext { + stdout: { write(text: string): unknown }; + stderr: { write(text: string): unknown }; + // Directory holding `session.json`. + sessionDir: string; + // The `--cache-dir` global, or undefined to let the library pick its + // per-user default keyed by the account id. + cacheDir?: string; + // Reads the session file into a client, or null when there is none. The + // CLI passes `loadSession` from `cli-session.ts`; tests pass a fake client. + loadSession: (path: string) => Client | null; + // Used by `login` only. The CLI passes `Client.login` and terminal + // prompts; tests pass fakes. + login: (opts: LoginOptions) => Promise<Client>; + prompt: (message: string) => Promise<string>; + // Like `prompt`, but the answer is masked as it is typed. + promptSecret: (message: string) => Promise<string>; +} + +const sessionPath = (ctx: CliContext): string => + join(ctx.sessionDir, "session.json"); + +// Write the session readable by its owner only, in a directory only its owner +// can enter. +export const saveSession = ( + sessionDir: string, + snapshot: ClientSnapshot, +): void => { + mkdirSync(sessionDir, { recursive: true, mode: 0o700 }); + writeFileSync( + join(sessionDir, "session.json"), + JSON.stringify(snapshot, null, 2), + { mode: 0o600 }, + ); +}; + +// The saved client, or undefined after telling the user why there is none. +const requireSession = (ctx: CliContext): Client | undefined => { + let client: Client | null; + try { + client = ctx.loadSession(sessionPath(ctx)); + } catch (err) { + ctx.stderr.write( + `${err instanceof Error ? err.message : err}\n` + + `Run "quak logout" and then "quak login" to replace it.\n`, + ); + return undefined; + } + if (!client) { + ctx.stderr.write( + `Not logged in. Run "quak login" first.\nSession file: ${sessionPath(ctx)}\n`, + ); + return undefined; + } + return client; +}; + +// A library client that omits `fetchMLData`, so the point commands below do not +// kick the library's background ML backfill: they read metadata, or fetch one +// file's content, and exit. `backup` and `backup-metadata` handle ML on their +// own terms. The content source is kept so `get`/`get-thumb`/`--exif` can fetch +// originals through the on-disk cache. +const readLibraryClient = (client: Client): LibraryClient => ({ + whoami: () => client.whoami(), + collectionsSince: (args) => client.collectionsSince(args), + filesSince: (args) => client.filesSince(args), + contentSource: () => client.contentSource(), +}); + +// Open a library for a single point command: the aggressive background precache +// (issue #48) is off — a one-shot `collections` or `get` must not start +// downloading the whole account — and the refresh interval is long so no second +// refresh fires mid-command. +const openReadLibrary = (ctx: CliContext, client: Client): Promise<Library> => + Library.open({ + client: readLibraryClient(client), + cacheDirectory: ctx.cacheDir, + refreshIntervalSeconds: 3600, + precacheThumbnails: false, + precacheOriginals: false, + }); + +export const loginCommand = async (ctx: CliContext): Promise<number> => { + await init(); + const email = process.env.QUAK_EMAIL ?? (await ctx.prompt("Email")); + const password = + process.env.QUAK_PASSWORD ?? (await ctx.promptSecret("Password")); + + ctx.stderr.write("Authenticating...\n"); + try { + const client = await ctx.login({ + email, + password, + totp: async () => ctx.prompt("TOTP code: "), + emailOTP: async () => ctx.prompt("Email verification code: "), + }); + + saveSession(ctx.sessionDir, client.toJSON()); + const info = client.whoami(); + ctx.stderr.write(`Logged in as ${info.email} (user ${info.userID})\n`); + ctx.stderr.write(`Session saved to ${sessionPath(ctx)}\n`); + } catch (err) { + ctx.stderr.write( + `Login failed: ${err instanceof Error ? err.message : err}\n`, + ); + return 1; + } + return 0; +}; + +export const whoamiCommand = async (ctx: CliContext): Promise<number> => { + await init(); + const client = requireSession(ctx); + if (!client) return 1; + const info = client.whoami(); + ctx.stdout.write(JSON.stringify(info) + "\n"); + return 0; +}; + +// Ends the session on the server, then deletes the session file even when that +// failed, and exits 1 if it did. The cache is left in place; the user is told +// where it is. +export const logoutCommand = async (ctx: CliContext): Promise<number> => { + const path = sessionPath(ctx); + if (!existsSync(path)) { + ctx.stderr.write("No session found.\n"); + return 0; + } + await init(); + let cacheDir = ctx.cacheDir; + let failure: string | undefined; + try { + const client = ctx.loadSession(path); + if (client) { + cacheDir ??= defaultCacheDirectory(client.whoami().userID); + await client.logoutOnServer(); + client.logout(); + } + } catch (err) { + failure = err instanceof Error ? err.message : String(err); + } + unlinkSync(path); + if (failure === undefined) { + ctx.stderr.write("Session ended on the server.\n"); + } else { + ctx.stderr.write( + `Could not end the session on the server: ${failure}\n`, + ); + } + ctx.stderr.write("Session deleted.\n"); + if (cacheDir !== undefined) { + ctx.stderr.write( + `Cache directory ${cacheDir} still holds decrypted data; delete it to remove that data.\n`, + ); + } + return failure === undefined ? 0 : 1; +}; + +export const collectionsCommand = async ( + ctx: CliContext, + opts: { json?: boolean }, +): Promise<number> => { + await init(); + const client = requireSession(ctx); + if (!client) return 1; + const lib = await openReadLibrary(ctx, client); + try { + // Force a server round-trip and list in enumeration order (issue #36 + // amendment, issue #52): the pre-library CLI printed current state in + // this order, not the albums projection's newest-first order. + const collections = await freshCollections(lib); + + if (opts.json) { + ctx.stdout.write( + JSON.stringify( + collections.map((c) => ({ + id: c.id, + name: c.name, + type: c.type, + ownerID: c.ownerID, + isShared: c.isShared, + updationTime: c.updationTime, + })), + null, + 2, + ) + "\n", + ); + } else { + for (const c of collections) { + ctx.stdout.write( + `${c.id}\t${c.type}\t${c.name}${c.isShared ? " (shared)" : ""}\n`, + ); + } + } + return 0; + } finally { + await lib.close(); + } +}; + +export const filesCommand = async ( + ctx: CliContext, + opts: { collection: string; json?: boolean }, +): Promise<number> => { + await init(); + const client = requireSession(ctx); + if (!client) return 1; + const collectionID = Number(opts.collection); + if (!Number.isFinite(collectionID)) { + ctx.stderr.write("Invalid collection ID\n"); + return 1; + } + + const lib = await openReadLibrary(ctx, client); + try { + // Force a server round-trip and list in enumeration order (issue #36 + // amendment, issue #52). Each file prints from its own decrypted + // metadata (raw title, microsecond creationTime) via cli-output, and in + // the pre-library CLI's enumeration order, not the projection's + // newest-first order. + const files = await freshFiles(lib, collectionID); + if (!files) { + ctx.stderr.write(`Collection ${collectionID} not found\n`); + return 1; + } + + if (opts.json) { + ctx.stdout.write( + JSON.stringify(files.map(fileListRow), null, 2) + "\n", + ); + } else { + for (const file of files) { + ctx.stdout.write(fileListLine(file) + "\n"); + } + } + return 0; + } finally { + await lib.close(); + } +}; + +export const getCommand = async ( + ctx: CliContext, + fileIDStr: string, + opts: { out?: string }, +): Promise<number> => { + await init(); + const client = requireSession(ctx); + if (!client) return 1; + const fileID = Number(fileIDStr); + if (!Number.isFinite(fileID)) { + ctx.stderr.write("Invalid file ID\n"); + return 1; + } + + const lib = await openReadLibrary(ctx, client); + try { + // Force a server round-trip so the file resolves against current state + // (issue #36 amendment, issue #52). + const resolved = await freshFile(lib, fileID); + if (!resolved) { + ctx.stderr.write(`File ${fileID} not found\n`); + return 1; + } + const { photo, file } = resolved; + + const result = await photo.original(); + // Default name is the file's own title, as the pre-library CLI used + // (not the editedName-preferring projection title) (issue #52). + const outPath = opts.out ?? originalName(file); + copyFileSync(result.path, outPath); + ctx.stderr.write(`${result.bytes} bytes -> ${outPath}\n`); + return 0; + } finally { + await lib.close(); + } +}; + +export const getThumbCommand = async ( + ctx: CliContext, + fileIDStr: string, + opts: { out?: string }, +): Promise<number> => { + await init(); + const client = requireSession(ctx); + if (!client) return 1; + const fileID = Number(fileIDStr); + if (!Number.isFinite(fileID)) { + ctx.stderr.write("Invalid file ID\n"); + return 1; + } + + const lib = await openReadLibrary(ctx, client); + try { + // Force a server round-trip so the file resolves against current state + // (issue #36 amendment, issue #52). + const resolved = await freshFile(lib, fileID); + if (!resolved) { + ctx.stderr.write(`File ${fileID} not found\n`); + return 1; + } + const { photo, file } = resolved; + + const result = await photo.thumbnail(); + // Default name is thumb_<file's own title>, as the pre-library CLI + // used (not the projection title) (issue #52). + const outPath = opts.out ?? thumbnailName(file); + copyFileSync(result.path, outPath); + ctx.stderr.write(`${result.bytes} bytes -> ${outPath}\n`); + return 0; + } finally { + await lib.close(); + } +}; + +export const backupMetadataCommand = async ( + ctx: CliContext, + dir: string, + opts: { exif?: boolean; all?: boolean }, +): Promise<number> => { + await init(); + const client = requireSession(ctx); + if (!client) return 1; + const lib = await openReadLibrary(ctx, client); + try { + // Refresh first so the dump holds current account state, not what the + // cache last held; a failed refresh throws. + await lib.fresh(); + const { failedMLBatches } = await runMetadataBackup(lib, client, dir, { + exif: opts.exif || opts.all, + onProgress: (msg) => ctx.stderr.write(msg + "\n"), + }); + return failedMLBatches > 0 ? 1 : 0; + } finally { + await lib.close(); + } +}; + +export const backupCommand = async ( + ctx: CliContext, + dir: string, + opts: { json?: boolean }, +): Promise<number> => { + await init(); + const client = requireSession(ctx); + if (!client) return 1; + + ctx.stderr.write("Starting backup...\n"); + // The precache is off: the backup fetches what it needs, and must not + // also fill the cache with every thumbnail and the recent originals. + const lib = await Library.open({ + client, + downloadDirectory: dir, + cacheDirectory: ctx.cacheDir, + precacheThumbnails: false, + precacheOriginals: false, + }); + try { + const result = await lib.backup({ + downloadDirectory: dir, + onProgress: (msg) => { + if (!opts.json) ctx.stderr.write(msg + "\n"); + }, + }); + + if (opts.json) { + ctx.stdout.write(JSON.stringify(result, null, 2) + "\n"); + } else { + ctx.stderr.write("\n--- Backup complete ---\n"); + ctx.stderr.write(` Total files: ${result.totalFiles}\n`); + ctx.stderr.write(` Downloaded: ${result.downloaded}\n`); + ctx.stderr.write(` Skipped: ${result.skipped}\n`); + ctx.stderr.write(` Failed: ${result.failed}\n`); + if (result.errors.length > 0) { + ctx.stderr.write("\nFailed files:\n"); + for (const e of result.errors) { + ctx.stderr.write( + ` [${e.collection}] ${e.title} (id ${e.fileID}): ${e.error}\n`, + ); + } + } + } + + return result.failed > 0 ? 1 : 0; + } finally { + await lib.close(); + } +}; + +export const listMissingThumbnailsCommand = async ( + ctx: CliContext, + opts: { json?: boolean }, +): Promise<number> => { + await init(); + const client = requireSession(ctx); + if (!client) return 1; + const lib = await openReadLibrary(ctx, client); + try { + // Refresh first so files added since the cache was written are + // checked; a failed refresh throws. + await lib.fresh(); + const missing = await listMissingThumbnails(lib, client, (msg) => { + if (!opts.json) ctx.stderr.write(msg + "\n"); + }); + + if (opts.json) { + ctx.stdout.write(JSON.stringify(missing, null, 2) + "\n"); + } else { + if (missing.length === 0) { + ctx.stderr.write("No missing thumbnails found.\n"); + } else { + ctx.stderr.write( + `\n${missing.length} file(s) with missing thumbnails:\n`, + ); + for (const m of missing) { + ctx.stdout.write( + `${m.fileID}\t${m.title}\t${m.collection}\t${m.reason}\n`, + ); + } + } + } + return 0; + } finally { + await lib.close(); + } +}; + +export const fixMissingThumbnailsCommand = async ( + ctx: CliContext, + opts: { file?: string[]; json?: boolean }, +): Promise<number> => { + await init(); + const client = requireSession(ctx); + if (!client) return 1; + const lib = await openReadLibrary(ctx, client); + try { + // Refresh first so files added since the cache was written are found; + // a failed refresh throws. + await lib.fresh(); + let fileIDs: number[]; + if (opts.file && opts.file.length > 0) { + fileIDs = opts.file.map(Number).filter(Number.isFinite); + } else { + ctx.stderr.write("Scanning for missing thumbnails...\n"); + const missing = await listMissingThumbnails(lib, client, (msg) => { + if (!opts.json) ctx.stderr.write(msg + "\n"); + }); + fileIDs = missing.map((m) => m.fileID); + if (fileIDs.length === 0) { + ctx.stderr.write("No missing thumbnails found.\n"); + return 0; + } + ctx.stderr.write(`Found ${fileIDs.length} file(s) to fix.\n`); + } + + const results = await fixMissingThumbnails( + lib, + client, + fileIDs, + (msg) => { + if (!opts.json) ctx.stderr.write(msg + "\n"); + }, + ); + + if (opts.json) { + ctx.stdout.write(JSON.stringify(results, null, 2) + "\n"); + } else { + const fixed = results.filter((r) => r.status === "fixed").length; + const skipped = results.filter( + (r) => r.status === "skipped", + ).length; + const failed = results.filter((r) => r.status === "failed").length; + ctx.stderr.write(`\n--- Done ---\n`); + ctx.stderr.write(` Fixed: ${fixed}\n`); + ctx.stderr.write(` Skipped: ${skipped}\n`); + ctx.stderr.write(` Failed: ${failed}\n`); + if (skipped > 0) { + ctx.stderr.write("\nSkipped:\n"); + for (const r of results.filter((r) => r.status === "skipped")) { + ctx.stderr.write( + ` ${r.fileID}\t${r.title}\t${r.reason}\n`, + ); + } + } + if (failed > 0) { + ctx.stderr.write("\nFailed files:\n"); + for (const r of results.filter((r) => r.status === "failed")) { + ctx.stderr.write( + ` ${r.fileID}\t${r.title}\t${r.reason}\n`, + ); + } + } + } + + return results.some((r) => r.status === "failed") ? 1 : 0; + } finally { + await lib.close(); + } +}; diff --git a/src/cli-output.ts b/src/cli-output.ts index c774221..2bf9b21 100644 --- a/src/cli-output.ts +++ b/src/cli-output.ts @@ -9,6 +9,7 @@ // `metadata.title`, and issue #52 requires that output stay byte-identical, so // the commands shape their output from the raw `EnteFile` through here. +import { sanitizeFileName } from "./filename.js"; import type { EnteFile, FileType, Microseconds } from "./model/types.js"; // One row of `quak files --json`. @@ -32,9 +33,11 @@ export const fileListRow = (file: EnteFile): FileListRow => ({ export const fileListLine = (file: EnteFile): string => `${file.id}\t${file.metadata.fileType}\t${file.metadata.title}`; -// Default output path for `quak get` when `--out` is not given. -export const originalName = (file: EnteFile): string => file.metadata.title; +// Default output path for `quak get` when `--out` is not given. The title comes +// from the server, so it is sanitized; `--out` is the user's and is used as is. +export const originalName = (file: EnteFile): string => + sanitizeFileName(file.metadata.title, `file-${file.id}`); // Default output path for `quak get-thumb` when `--out` is not given. export const thumbnailName = (file: EnteFile): string => - `thumb_${file.metadata.title}`; + `thumb_${originalName(file)}`; diff --git a/src/cli-run.ts b/src/cli-run.ts new file mode 100644 index 0000000..c668b49 --- /dev/null +++ b/src/cli-run.ts @@ -0,0 +1,36 @@ +// Runs one CLI command for `bin/quak.ts` and exits with its code. + +import type { Writable } from "node:stream"; + +// Run a command and exit with its code once stdout/stderr have drained. +// Exiting before the drain can truncate piped output, and the library can keep +// the event loop alive after a command returns, so a plain return could hang. +// An error the command throws is printed as one `quak: MESSAGE` line, without +// the stack trace, and exits 1. +export const run = async ( + command: Promise<number>, + stdout: Writable, + stderr: Writable, + exit: (code: number) => void, +): Promise<void> => { + let code: number; + try { + code = await command; + } catch (err) { + stderr.write( + `quak: ${err instanceof Error ? err.message : String(err)}\n`, + ); + code = 1; + } + const pending = [stdout, stderr].filter((s) => s.writableLength > 0); + if (pending.length === 0) { + exit(code); + return; + } + let remaining = pending.length; + for (const s of pending) { + s.once("drain", () => { + if (--remaining === 0) exit(code); + }); + } +}; diff --git a/src/cli-session.ts b/src/cli-session.ts new file mode 100644 index 0000000..72228d2 --- /dev/null +++ b/src/cli-session.ts @@ -0,0 +1,26 @@ +// How the CLI reads its saved session file back into a `Client`. +// +// A missing file means "not logged in" and returns null. A file that exists but +// cannot be read back into a client (bad JSON, a missing field, a key of the +// wrong length) throws an error saying the session file is corrupt, so the CLI +// can tell the user which of the two it is. Needs `init()` first. + +import { existsSync, readFileSync } from "node:fs"; +import type { ApiClientOptions } from "./api/client.js"; +import { Client } from "./client.js"; + +export const loadSession = ( + path: string, + apiOptions?: ApiClientOptions, +): Client | null => { + if (!existsSync(path)) return null; + try { + return Client.fromJSON( + JSON.parse(readFileSync(path, "utf-8")), + apiOptions, + ); + } catch (err) { + const reason = err instanceof Error ? err.message : String(err); + throw new Error(`Session file ${path} is corrupt: ${reason}`); + } +}; diff --git a/src/client.ts b/src/client.ts index 3acfd48..62784c1 100644 --- a/src/client.ts +++ b/src/client.ts @@ -125,18 +125,57 @@ export class Client { ); } - static fromJSON( - snapshot: ClientSnapshot, - apiOptions?: ApiClientOptions, - ): Client { - const api = new ApiClient({ ...apiOptions, authToken: snapshot.token }); + // Restore a client from a `toJSON()` snapshot. The snapshot usually comes + // straight from `JSON.parse` of a file on disk, so every field is checked + // before use; a bad one throws an error naming it. Needs `init()` first. + static fromJSON(snapshot: unknown, apiOptions?: ApiClientOptions): Client { + const invalid = (field: string, problem: string): Error => + new Error(`Invalid session data: ${field} ${problem}`); + + if (typeof snapshot !== "object" || snapshot === null) { + throw new Error("Invalid session data: not a JSON object"); + } + const s = snapshot as Record<string, unknown>; + for (const field of ["email", "token"]) { + if (typeof s[field] !== "string" || s[field] === "") { + throw invalid(field, "must be a non-empty string"); + } + } + if (!Number.isInteger(s.userID)) { + throw invalid("userID", "must be an integer"); + } + const key = (field: string): Uint8Array => { + const value = s[field]; + if (typeof value !== "string") { + throw invalid(field, "must be a base64 string"); + } + let bytes: Uint8Array; + try { + bytes = fromBase64(value); + } catch { + throw invalid(field, "is not valid base64"); + } + // The master key (secretbox) and the key pair (box) are all 32 bytes. + if (bytes.length !== 32) { + throw invalid( + field, + `must decode to 32 bytes, got ${bytes.length}`, + ); + } + return bytes; + }; + + const api = new ApiClient({ + ...apiOptions, + authToken: s.token as string, + }); return new Client( api, - snapshot.email, - snapshot.userID, - fromBase64(snapshot.masterKey), - fromBase64(snapshot.secretKey), - fromBase64(snapshot.publicKey), + s.email as string, + s.userID as number, + key("masterKey"), + key("secretKey"), + key("publicKey"), ); } @@ -164,19 +203,37 @@ export class Client { toJSON(): ClientSnapshot { this.assertLoggedIn(); + const token = this.api.getAuthToken(); + if (!token) { + throw new Error("Cannot serialize client: it has no auth token"); + } return { email: this.email, userID: this.userID, - token: this.api["token"]!, + token, masterKey: toBase64(this.masterKey), secretKey: toBase64(this.secretKey), publicKey: toBase64(this.publicKey), }; } + // Ends this client's session on the server (`POST /users/logout`), so the + // token stops working everywhere, including in any saved copy of it. This + // client is left as it was; call `logout()` to clear it. + async logoutOnServer(): Promise<void> { + this.assertLoggedIn(); + await this.api.postJSON("/users/logout", {}); + } + + // Zeroes the key buffers in place, so any copy of the reference held + // elsewhere is wiped too. Every method checks `assertLoggedIn` before + // touching the keys, so nothing decrypts with the zeroed keys. logout(): void { this.loggedOut = true; this.api.clearAuthToken(); + this.masterKey.fill(0); + this.secretKey.fill(0); + this.publicKey.fill(0); } // Enumerate collections changed since `sinceTime`. Live collections are @@ -192,6 +249,8 @@ export class Client { const { collections: raws } = await this.api.getJSON<{ collections: RawCollection[]; }>("/collections/v2", { sinceTime: args.sinceTime }); + // logout() may have zeroed the keys while the request was in flight. + this.assertLoggedIn(); const collections: Collection[] = []; const deleted: number[] = []; diff --git a/src/crypto/hash.ts b/src/crypto/hash.ts new file mode 100644 index 0000000..b94d94f --- /dev/null +++ b/src/crypto/hash.ts @@ -0,0 +1,22 @@ +import sodium, { type StateAddress } from "libsodium-wrappers-sumo"; +import { toBase64 } from "./encoding.js"; + +// The content hash an uploading client records in a file's metadata: unkeyed +// BLAKE2b with a 64-byte output over the original's bytes, fed in chunks, as +// standard base64 with padding. Named after the upstream client's functions. +// The output length is read at call time for the same reason as +// `streamTagFinal` in stream.ts: libsodium sets its constants only once ready. + +export const chunkHashInit = (): StateAddress => + sodium.crypto_generichash_init(null, sodium.crypto_generichash_BYTES_MAX); + +export const chunkHashUpdate = (state: StateAddress, chunk: Uint8Array): void => + sodium.crypto_generichash_update(state, chunk); + +export const chunkHashFinal = (state: StateAddress): string => + toBase64( + sodium.crypto_generichash_final( + state, + sodium.crypto_generichash_BYTES_MAX, + ), + ); diff --git a/src/crypto/index.ts b/src/crypto/index.ts index 4e3e53b..d93af09 100644 --- a/src/crypto/index.ts +++ b/src/crypto/index.ts @@ -7,6 +7,7 @@ export { } from "./encoding.js"; export { deriveKEK, deriveLoginSubkey } from "./kdf.js"; export { decryptBox, decryptSealed } from "./box.js"; +export { chunkHashFinal, chunkHashInit, chunkHashUpdate } from "./hash.js"; export { decryptBlob, encryptBlob, diff --git a/src/download/index.ts b/src/download/index.ts index 19e17f7..06cd18a 100644 --- a/src/download/index.ts +++ b/src/download/index.ts @@ -1,8 +1,13 @@ -import { randomUUID } from "node:crypto"; +import { randomBytes } from "node:crypto"; +import { readdirSync, rmSync } from "node:fs"; import { open, rename, rm } from "node:fs/promises"; import type { FileHandle } from "node:fs/promises"; import { dirname, join } from "node:path"; +import { Unzip, UnzipInflate } from "fflate"; import { + chunkHashFinal, + chunkHashInit, + chunkHashUpdate, fromBase64, initStreamPull, pullStreamChunk, @@ -11,6 +16,7 @@ import { streamTagFinal, } from "../crypto/index.js"; import { TruncatedStreamError } from "../errors.js"; +import { sanitizeFileName } from "../filename.js"; import { withRetry } from "../retry.js"; import type { ApiClient } from "../api/client.js"; import type { EnteFile } from "../model/types.js"; @@ -97,47 +103,53 @@ const streamDecrypt = async ( onProgress?.(totalPlain); }; - for (;;) { - const { done, value } = await reader.read(); - if (value && value.length > 0) { - pending.push(value); - pendingBytes += value.length; - } + try { + for (;;) { + const { done, value } = await reader.read(); + if (value && value.length > 0) { + pending.push(value); + pendingBytes += value.length; + } - while (pendingBytes >= ENC_CHUNK_SIZE) { - const encChunk = takeContiguous(ENC_CHUNK_SIZE); - // A whole chunk that fails to authenticate while the stream carries - // on is corruption, not truncation; that error propagates unchanged. - const { plaintext, tag } = pullStreamChunk(state, encChunk); - await consume(plaintext, tag); - } + while (pendingBytes >= ENC_CHUNK_SIZE) { + const encChunk = takeContiguous(ENC_CHUNK_SIZE); + // A whole chunk that fails to authenticate while the stream + // carries on is corruption, not truncation; that error + // propagates unchanged. + const { plaintext, tag } = pullStreamChunk(state, encChunk); + await consume(plaintext, tag); + } - if (done) { - if (pendingBytes > 0) { - const buffer = takeContiguous(pendingBytes); - // Whatever is left over once every whole chunk has been - // consumed must be the stream's final chunk, and a final - // chunk that actually arrived in full authenticates. If it - // does not, the body stopped part-way through a chunk — the - // ordinary shape of a dropped connection. Poly1305 cannot - // tell a partial chunk from a corrupt one, so this is - // reported as the truncation it almost always is, with the - // authentication failure kept as the error's cause. Only the - // pull is guarded: a sink failure on a chunk that did - // authenticate is a disk error, not a truncation. - let pulled; - try { - pulled = pullStreamChunk(state, buffer); - } catch (err) { - throw new TruncatedStreamError( - `download: stream truncated: response body ended with ${buffer.length} trailing bytes that did not authenticate as a final chunk (transfer stopped mid-chunk, or the data is corrupt)`, - { cause: err }, - ); + if (done) { + if (pendingBytes > 0) { + const buffer = takeContiguous(pendingBytes); + // Whatever is left over once every whole chunk has been + // consumed must be the stream's final chunk, and a final + // chunk that actually arrived in full authenticates. If + // it does not, the body stopped part-way through a chunk + // — the ordinary shape of a dropped connection. Poly1305 + // cannot tell a partial chunk from a corrupt one, so this + // is reported as the truncation it almost always is, with + // the authentication failure kept as the error's cause. + // Only the pull is guarded: a sink failure on a chunk + // that did authenticate is a disk error, not a + // truncation. + let pulled; + try { + pulled = pullStreamChunk(state, buffer); + } catch (err) { + throw new TruncatedStreamError( + `download: stream truncated: response body ended with ${buffer.length} trailing bytes that did not authenticate as a final chunk (transfer stopped mid-chunk, or the data is corrupt)`, + { cause: err }, + ); + } + await consume(pulled.plaintext, pulled.tag); } - await consume(pulled.plaintext, pulled.tag); + break; } - break; } + } finally { + reader.releaseLock(); } // Only the last chunk of a secretstream carries TAG_FINAL. Everything a @@ -157,6 +169,50 @@ const streamDecrypt = async ( return totalPlain; }; +// Fsync a file or a directory, so its contents (for a directory, its entries) +// are on stable storage. Exported for the backup tree's copy, which needs the +// same durability as the writer below. +export const fsyncPath = async (path: string): Promise<void> => { + const handle = await open(path, "r"); + try { + await handle.sync(); + } finally { + await handle.close(); + } +}; + +// A process-ID check: signal 0 delivers nothing and only reports whether the +// process exists. EPERM means it exists but belongs to another user. +const isRunning = (pid: number): boolean => { + try { + process.kill(pid, 0); + return true; + } catch (err) { + return (err as NodeJS.ErrnoException).code === "EPERM"; + } +}; + +// Delete the temp files a killed process left in `dir`: the writer's +// `.quak-<pid>-<random>.tmp` and the backup copy's +// `.quak-backup-<name>-<pid>-<random>.tmp`. Only files whose process is no +// longer running are removed, so another process writing into the same +// directory keeps its own. A reused process ID can only keep a leftover a while +// longer, never remove a live one. +export const removeLeftoverTempFiles = (dir: string): void => { + let names: string[]; + try { + names = readdirSync(dir); + } catch { + return; + } + for (const name of names) { + const match = /^\.quak-(?:.*-)?(\d+)-[0-9a-z]*\.tmp$/.exec(name); + if (match && !isRunning(Number(match[1]))) { + rmSync(join(dir, name), { force: true }); + } + } +}; + // Stage a write to `destination` atomically and durably, then rename it into // place. `fill` writes the contents into the open temp file handle — either the // whole buffer at once (`writeAtomic`) or chunk by chunk as they decrypt @@ -182,8 +238,12 @@ const stageAtomic = async ( ): Promise<void> => { const dir = dirname(destination); // The random suffix keeps concurrent downloads of the same destination - // from stepping on each other's temporary file. - const tmpPath = join(dir, `.quak-${randomUUID()}.tmp`); + // from stepping on each other's temporary file; the process ID lets + // `removeLeftoverTempFiles` tell a leftover from a write in progress. + const tmpPath = join( + dir, + `.quak-${process.pid}-${randomBytes(16).toString("hex")}.tmp`, + ); try { const handle = await open(tmpPath, "w"); try { @@ -192,16 +252,15 @@ const stageAtomic = async ( } finally { await handle.close(); } + // `rename` replaces the destination's directory entry rather than + // writing through it: an existing symlink at `destination` is + // replaced, not followed, and the new file has the temp file's + // permissions, not those of the file it replaced. await rename(tmpPath, destination); // Fsync the directory so the rename itself survives a crash: renaming // over a synced temp file still leaves the new directory entry in the // page cache until the directory is synced. - const dirHandle = await open(dir, "r"); - try { - await dirHandle.sync(); - } finally { - await dirHandle.close(); - } + await fsyncPath(dir); } catch (err) { // Best-effort cleanup. A failure to remove the temporary file must // never replace the error that actually explains what went wrong. @@ -219,31 +278,139 @@ export const writeAtomic = async ( ): Promise<void> => stageAtomic(destination, (handle) => handle.writeFile(plaintext)); +// Hashes an original's bytes as they are decrypted, for comparison with the +// hash its uploader recorded. +interface ContentHasher { + update: (plaintext: Uint8Array) => void; + digest: () => string; +} + +const fileHasher = (): ContentHasher => { + const state = chunkHashInit(); + return { + update: (plaintext) => chunkHashUpdate(state, plaintext), + digest: () => chunkHashFinal(state), + }; +}; + +// A live photo is stored as a ZIP of its image and its video, and its recorded +// hash is `<imageHash>:<videoHash>`, each over that part's own bytes. Like the +// upstream client's decoder, this takes the first entries whose names start +// with `image` and `video`. +// +// The ZIP is chosen by its uploader and may expand enormously, so entries are +// hashed as they decompress and never held. fflate's `Unzip` inflates each +// push in one piece, and deflate expands at most about 1000-fold, so the ZIP +// is pushed in 4 KiB slices to keep each decompressed piece near 4 MiB, one +// plaintext chunk. Every entry is started, even one that is not hashed, +// because fflate keeps an unstarted entry's data in memory. +const livePhotoHasher = (fileID: number): ContentHasher => { + const sliceSize = 4096; + const fail = (message: string, cause?: unknown): Error => + new Error(`download: file ${fileID}: ${message}`, { cause }); + const claimed = new Set<string>(); + const hashes = new Map<string, string>(); + const unzip = new Unzip((entry) => { + const part = ["image", "video"].find((p) => entry.name.startsWith(p)); + const target = + part === undefined || claimed.has(part) + ? undefined + : { part, state: chunkHashInit() }; + if (target !== undefined) claimed.add(target.part); + entry.ondata = (err, data, final) => { + if (err) throw err; + if (target === undefined) return; + chunkHashUpdate(target.state, data); + if (final) hashes.set(target.part, chunkHashFinal(target.state)); + }; + entry.start(); + }); + unzip.register(UnzipInflate); + // fflate reports a bad ZIP by throwing, sometimes a TypeError, which the + // retry would take for a network failure; a bad ZIP is never retried. + const push = (data: Uint8Array, final: boolean): void => { + try { + unzip.push(data, final); + } catch (err) { + throw fail("live photo is not a readable ZIP", err); + } + }; + return { + update: (plaintext) => { + for (let i = 0; i < plaintext.length; i += sliceSize) { + push(plaintext.subarray(i, i + sliceSize), false); + } + }, + digest: () => { + push(new Uint8Array(0), true); + const image = hashes.get("image"); + const video = hashes.get("video"); + if (image === undefined || video === undefined) { + throw fail( + "live photo ZIP does not hold both an image and a video", + ); + } + return `${image}:${video}`; + }, + }; +}; + // Decrypt `stream` straight to `destination`, one plaintext chunk at a time, // under the atomic writer's temp-then-rename discipline. Memory stays bounded // by the chunk size: each decrypted chunk is written to the temp file and // dropped. The rename happens only after the stream authenticates as terminated // on TAG_FINAL; a truncated stream throws and leaves the destination untouched. // Returns the plaintext length written. +// +// `original` is the file whose original this is (none for a thumbnail, which +// has no recorded hash). When its metadata has a hash, the decrypted bytes +// must match it or nothing is stored. Both a plain file and a live photo's +// parts are hashed as they stream. The mismatch error is not retried. const decryptToTemp = async ( destination: string, stream: ReadableStream<Uint8Array>, header: Uint8Array, key: Uint8Array, onProgress?: ProgressCallback, + original?: EnteFile, ): Promise<number> => { + const expected = original?.metadata.hash; + const hasher = + original === undefined || expected === undefined + ? undefined + : original.metadata.fileType === "livePhoto" + ? livePhotoHasher(original.id) + : fileHasher(); let bytesWritten = 0; - await stageAtomic(destination, async (handle) => { - bytesWritten = await streamDecrypt( - stream, - header, - key, - async (plaintext) => { - await handle.write(plaintext); - }, - onProgress, - ); - }); + try { + await stageAtomic(destination, async (handle) => { + bytesWritten = await streamDecrypt( + stream, + header, + key, + async (plaintext) => { + hasher?.update(plaintext); + await handle.write(plaintext); + }, + onProgress, + ); + if (original === undefined || hasher === undefined) return; + const actual = hasher.digest(); + if (actual !== expected) { + throw new Error( + `download: file ${original.id}: content hash ${actual} does not match the hash its uploader recorded, ${expected}`, + ); + } + }); + } catch (err) { + // Cancel the body so its connection is closed now rather than held + // until the stream is garbage collected. A backup run carries on past + // a failed file, so without this every failure would hold a socket. + // This covers every failure, including a temp file that cannot be + // opened and a header that is rejected before the body is read. + await stream.cancel(err).catch(() => undefined); + throw err; + } return bytesWritten; }; @@ -274,10 +441,18 @@ const fetchAndDecrypt = async ( key: Uint8Array, destination: string, onProgress?: ProgressCallback, + original?: EnteFile, ): Promise<number> => withRetry(async () => { const stream = await openStream(); - return decryptToTemp(destination, stream, header, key, onProgress); + return decryptToTemp( + destination, + stream, + header, + key, + onProgress, + original, + ); }, api.getRetryOptions()); export const downloadFile = async ( @@ -286,7 +461,10 @@ export const downloadFile = async ( outPath?: string, onProgress?: ProgressCallback, ): Promise<DownloadResult> => { - const resolvedPath = outPath ?? file.metadata.title; + // `outPath` is the caller's and is used as is; the title is the server's + // and is sanitized so it can only name a file in the current directory. + const resolvedPath = + outPath ?? sanitizeFileName(file.metadata.title, `file-${file.id}`); const header = fromBase64(file.file.decryptionHeader); const bytesWritten = await fetchAndDecrypt( api, @@ -295,6 +473,7 @@ export const downloadFile = async ( file.key, resolvedPath, onProgress, + file, ); return { path: resolvedPath, bytesWritten }; }; @@ -305,7 +484,9 @@ export const downloadThumbnail = async ( outPath?: string, onProgress?: ProgressCallback, ): Promise<DownloadResult> => { - const resolvedPath = outPath ?? `thumb_${file.metadata.title}`; + const resolvedPath = + outPath ?? + `thumb_${sanitizeFileName(file.metadata.title, `file-${file.id}`)}`; const header = fromBase64(file.thumbnail.decryptionHeader); const bytesWritten = await fetchAndDecrypt( api, diff --git a/src/filename.ts b/src/filename.ts new file mode 100644 index 0000000..963257f --- /dev/null +++ b/src/filename.ts @@ -0,0 +1,37 @@ +// File names built from server-supplied metadata. +// +// A file's title and a collection's name are decrypted from data the server +// hands us, and quak does not trust the server. Any name taken from them and +// used on disk goes through here, so it can only ever name one file inside the +// directory the caller chose: never a path, never `..`, never hidden, never a +// Windows device name. +// +// A path the user typed (`--out`, `outPath`) is not passed through here: the +// caller is trusted, the server is not. + +import { extname } from "node:path"; + +// Path separators, characters Windows forbids in file names, and control +// characters (NUL included). +// eslint-disable-next-line no-control-regex +const UNSAFE_CHARACTERS = /[/\\:*?"<>|\x00-\x1f\x7f]/g; + +// Names Windows reserves for devices, with or without an extension. +const RESERVED_DEVICE_NAME = /^(con|prn|aux|nul|com[1-9]|lpt[1-9])(\.|$)/i; + +// `name` made safe to use as a single file name. Each unsafe character becomes +// `_`, a leading run of dots becomes one `_`, and a device name gets a leading +// `_`. A name with none of these comes back unchanged. An empty name becomes +// `fallback`, which the caller derives from the record's ID. +export const sanitizeFileName = (name: string, fallback: string): string => { + if (name === "") return fallback; + const cleaned = name.replace(UNSAFE_CHARACTERS, "_").replace(/^\.+/, "_"); + return RESERVED_DEVICE_NAME.test(cleaned) ? `_${cleaned}` : cleaned; +}; + +// The extension of `title` (".jpg"), or ".bin" when it has none or it holds +// anything but letters and digits. +export const safeExtension = (title: string): string => { + const ext = extname(title); + return /^\.[A-Za-z0-9]+$/.test(ext) ? ext : ".bin"; +}; diff --git a/src/index.ts b/src/index.ts index 438e60e..a0a4186 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1,4 +1,8 @@ -export const VERSION = "0.0.0"; +// package.json is the one place the version is written. tsc copies it to +// dist/package.json, so this path resolves from source and from dist/src/. +import pkg from "../package.json" with { type: "json" }; + +export const VERSION: string = pkg.version; export { Client, diff --git a/src/library/content.ts b/src/library/content.ts index 2cb19d4..fce4c9b 100644 --- a/src/library/content.ts +++ b/src/library/content.ts @@ -15,12 +15,12 @@ // Integrity. The reused streaming decrypt is the enforced guarantee: every // chunk is authenticated and the writer renames the file into place only once // the stream ends on TAG_FINAL, so a truncated or corrupt fetch throws and -// nothing is stored. On top of that this module refuses to record a stored file -// that came out empty. The design also asks for a content-hash comparison -// against `FileMetadata.hash` (with a `fileSize` fallback); that is deferred — -// see the PR — because the exact hash construction cannot be confirmed against -// the repo's fixtures and `FileBlob.size` is the encrypted object size, not the -// decrypted length this layer has. +// nothing is stored. For an original whose metadata records a content hash +// (`FileMetadata.hash`), the writer also hashes the decrypted bytes and stores +// nothing if they differ, failing the fetch with an error naming the file. An +// original with no recorded hash is stored unchecked, as the upstream client +// does; thumbnails have none. On top of that this module refuses to record a +// stored file that came out empty. import { existsSync, statSync } from "node:fs"; import { @@ -39,14 +39,14 @@ import { downloadFile, downloadThumbnail, type ProgressCallback, + removeLeftoverTempFiles, } from "../download/index.js"; +import { safeExtension } from "../filename.js"; import type { EnteFile } from "../model/types.js"; import type { Priority, RequestPools } from "./pools.js"; const DIR_MODE = 0o700; const FILE_MODE = 0o600; -const TEMP_PREFIX = ".quak-"; -const TEMP_SUFFIX = ".tmp"; const GIB = 1024 * 1024 * 1024; // Owner ruling (#36): bound the originals cache at 100 GiB, but back off when // the volume has under 50 GiB free so the cache never crowds the disk. @@ -209,10 +209,8 @@ class AbortDrop extends Error { } } -const originalName = (file: EnteFile): string => { - const ext = extname(file.metadata.title || "") || ".bin"; - return `${file.id}${ext}`; -}; +const originalName = (file: EnteFile): string => + `${file.id}${safeExtension(file.metadata.title)}`; // The fileID a cache filename encodes, or undefined when the name is not one // the cache writes (`<digits><ext>`). @@ -323,6 +321,23 @@ export class ContentCache implements PhotoContent, ThumbnailsAPI { return this.get(fileID, "thumbnail", "on-demand", opts?.onProgress); } + // Get an original for a backup. One not present anywhere is written + // straight to `destination` and recorded there, so no second copy lands + // in the cache; one already present is returned where it is. + async backupOriginal( + fileID: number, + destination: string, + ): Promise<ContentResult> { + const result = await this.acquire( + fileID, + "original", + "on-demand", + undefined, + { destination }, + ); + return { path: result.path, bytes: result.bytes }; + } + async ensure(args: EnsureOptions): Promise<EnsureResult[]> { return this.ensureThumbnails(args); } @@ -424,13 +439,14 @@ export class ContentCache implements PhotoContent, ThumbnailsAPI { // The core: return the cached path if present, else fetch through the pool, // store, and return it. `cached` distinguishes a present hit (no network, - // no download event) from a fresh fetch. + // no download event) from a fresh fetch. A fetched original is stored at + // `opts.destination` when given, instead of in `originalsDir`. private async acquire( fileID: number, kind: Kind, priority: Priority, signal: AbortSignal | undefined, - opts?: { onByte?: ProgressCallback }, + opts?: { onByte?: ProgressCallback; destination?: string }, ): Promise<{ path: string; bytes: number; cached: boolean }> { const file = this.getFile(fileID); if (!file) throw new Error(`content cache: unknown file ${fileID}`); @@ -471,7 +487,7 @@ export class ContentCache implements PhotoContent, ThumbnailsAPI { kind === "original" ? this.originalsDir : this.thumbnailsDir; const dest = kind === "original" - ? join(dir, originalName(file)) + ? (opts?.destination ?? join(dir, originalName(file))) : join(dir, `${fileID}${THUMBNAIL_EXT}`); const pool = kind === "original" ? this.pools.content : this.pools.thumbnails; @@ -655,6 +671,9 @@ export class ContentCache implements PhotoContent, ThumbnailsAPI { } private async scan(dir: string, into: Map<number, string>): Promise<void> { + // Another process sharing this cache may still be writing its temp + // files, so only those whose process has exited are removed. + removeLeftoverTempFiles(dir); let entries: string[]; try { entries = await readdir(dir); @@ -662,12 +681,6 @@ export class ContentCache implements PhotoContent, ThumbnailsAPI { return; } for (const name of entries) { - if (name.startsWith(TEMP_PREFIX) && name.endsWith(TEMP_SUFFIX)) { - await rm(join(dir, name), { force: true }).catch( - () => undefined, - ); - continue; - } const id = fileIDFromName(name); const path = join(dir, name); if (id !== undefined && existsSync(path)) into.set(id, path); diff --git a/src/library/index.ts b/src/library/index.ts index 3b184cb..a7d7865 100644 --- a/src/library/index.ts +++ b/src/library/index.ts @@ -26,6 +26,7 @@ // store marked unsaved until a later save actually lands, so a stuck disk is // never masked by a subsequent empty refresh. +import { rm } from "node:fs/promises"; import { join } from "node:path"; import envPaths from "env-paths"; @@ -97,6 +98,11 @@ export { export const DEFAULT_REFRESH_INTERVAL_SECONDS = 3; +// The account's cache directory when `cacheDirectory` is not given: the +// env-paths cache directory plus the user id, so each account has its own. +export const defaultCacheDirectory = (userID: number): string => + join(envPaths("quak", { suffix: "" }).cache, String(userID)); + // Project a metadata store into by-id records, filling each record's cache // paths from the content cache when one is given. Shared by the live read // projection and the precache's initial seeding at open(). @@ -266,8 +272,9 @@ export class Library { // that, a fresh read propagates it. private cycle?: Promise<void>; // Guards the ML fetch pass so a slow backfill never runs twice at once; a - // refresh whose pass is still running kicks nothing new. - private mlFetching = false; + // refresh whose pass is still running kicks nothing new. Holds the running + // pass, so `close()` can wait for it. + private mlFetch?: Promise<void>; private closed = false; private lastRefreshAt?: number; private lastError?: string; @@ -342,11 +349,21 @@ export class Library { static async open(opts: LibraryOptions): Promise<Library> { const { userID } = opts.client.whoami(); const cacheDirectory = - opts.cacheDirectory ?? - join(envPaths("quak", { suffix: "" }).cache, String(userID)); - const store = await MetadataStore.load( - join(cacheDirectory, "metadata.json"), - ); + opts.cacheDirectory ?? defaultCacheDirectory(userID); + const metadataPath = join(cacheDirectory, "metadata.json"); + let store = await MetadataStore.load(metadataPath); + // A cache directory given explicitly can hold another account's cache. + // Its records and cursor are not this account's, so delete it and the + // ML data beside it and start empty. A user ID of 0 means the cache + // was never refreshed and so holds nothing to discard. + if (store.userID !== 0 && store.userID !== userID) { + await rm(metadataPath, { force: true }); + await rm(join(cacheDirectory, "mldata"), { + recursive: true, + force: true, + }); + store = await MetadataStore.load(metadataPath); + } const intervalMs = (opts.refreshIntervalSeconds ?? DEFAULT_REFRESH_INTERVAL_SECONDS) * 1000; @@ -521,11 +538,13 @@ export class Library { } // Back up every in-scope file to `downloadDirectory` in the historical - // on-disk layout, with a durable failure ledger (issue #51). Refreshes - // first, fetches pending originals (and optional thumbnails) through the - // content cache and pools, then rebuilds the derived symlink/JSON views - // from the model. Throws before any network work when no download directory - // is available or no content cache backs the originals it must fetch. + // on-disk layout, with a durable failure ledger (issue #51). Waits for a + // completed refresh first, as `fresh()` does, joining one already running, + // and rejects before touching any file when it fails. Then fetches pending + // originals (and optional thumbnails) through the content cache and pools, + // and rebuilds the derived symlink/JSON views from the model. Throws before + // any network work when no download directory is available or no content + // cache backs the originals it must fetch. backup(opts?: BackupOptions): Promise<BackupResult> { const downloadDirectory = opts?.downloadDirectory ?? this.downloadDirectory; @@ -549,10 +568,11 @@ export class Library { const cache = this.cache; return runBackup( { - refresh: () => this.runRefresh(), + refresh: () => this.refreshNow(), listCollections: () => this.store.listCollections(), listFiles: (id) => this.store.listFiles(id), - original: (fileID) => cache!.original(fileID), + original: (fileID, destination) => + cache!.backupOriginal(fileID, destination), thumbnail: (fileID) => cache!.thumbnail(fileID), }, { ...opts, downloadDirectory }, @@ -560,14 +580,21 @@ export class Library { } // Stop the background timer. Idempotent. An in-flight refresh is left to - // finish; it will not schedule another cycle once closed. - close(): void { + // finish; it will not schedule another cycle once closed. The returned + // promise resolves once that refresh (including its cache write), the ML + // fetch pass and the precache fetches already running have all finished, + // so a caller can then remove the cache directory. A refresh failure is + // reported through `status()`, not thrown here. + async close(): Promise<void> { this.closed = true; - this.precache?.close(); + const precacheClosed = this.precache?.close(); if (this.timer !== undefined) { clearTimeout(this.timer); this.timer = undefined; } + await this.cycle?.catch(() => {}); + await this.mlFetch; + await precacheClosed; } private scheduleNext(): void { @@ -631,7 +658,9 @@ export class Library { // outside the refresh's success/failure so a fetch or disk problem // there never marks the metadata refresh failed, and it is not // awaited so it never stalls the refresh interval. - void this.runMLFetch(); + this.mlFetch ??= this.runMLFetch().finally(() => { + this.mlFetch = undefined; + }); } catch (err) { const error = err instanceof Error ? err.message : String(err); this.lastError = error; @@ -745,13 +774,12 @@ export class Library { // Bind so the call keeps the client as its receiver when invoked // through the pool below. const fetchMLData = this.client.fetchMLData?.bind(this.client); - if (!mldata || !fetchMLData || this.closed || this.mlFetching) return; + if (!mldata || !fetchMLData || this.closed) return; const files = this.uniqueFiles(); const needed = mldata.neededFor(files); if (needed.length === 0) return; - this.mlFetching = true; this.emit({ operation: "fetchMLData", status: "started" }); try { const fileKeys = new Map<number, Uint8Array>(); @@ -784,8 +812,6 @@ export class Library { const error = err instanceof Error ? err.message : String(err); this.lastMLError = error; this.emit({ operation: "fetchMLData", status: "failed", error }); - } finally { - this.mlFetching = false; } } diff --git a/src/library/precache.ts b/src/library/precache.ts index f604eaa..472d627 100644 --- a/src/library/precache.ts +++ b/src/library/precache.ts @@ -92,9 +92,10 @@ export class Precache { private pinned = new Set<number>(); // A sweep runs at most once per fill at a time; a re-kick while one runs is - // a no-op, and the next refresh re-kicks after it finishes. - private thumbRunning = false; - private originalsRunning = false; + // a no-op, and the next refresh re-kicks after it finishes. Each holds the + // running sweep, so `close()` can wait for it. + private thumbSweep?: Promise<void>; + private originalsSweep?: Promise<void>; private readonly aborter = new AbortController(); private closed = false; @@ -191,15 +192,19 @@ export class Precache { } // Stop the fills. In-flight fetches are left to settle; queued ones drop. - close(): void { + // Resolves once both sweeps have finished, so nothing is still writing. + async close(): Promise<void> { this.closed = true; this.aborter.abort(); + await Promise.all([ + this.thumbSweep?.catch(() => {}), + this.originalsSweep?.catch(() => {}), + ]); } private kickThumbnails(): void { - if (this.thumbRunning) return; - this.thumbRunning = true; - void this.sweep( + if (this.thumbSweep) return; + this.thumbSweep = this.sweep( "precacheThumbnails", () => this.thumbOrder, (id) => this.cache!.pathsFor(id).thumbnailPath !== undefined, @@ -211,14 +216,13 @@ export class Precache { signal: this.aborter.signal, }), ).finally(() => { - this.thumbRunning = false; + this.thumbSweep = undefined; }); } private kickOriginals(): void { - if (this.originalsRunning) return; - this.originalsRunning = true; - void this.sweep( + if (this.originalsSweep) return; + this.originalsSweep = this.sweep( "precacheOriginals", () => this.originalsOrder, (id) => this.cache!.pathsFor(id).originalPath !== undefined, @@ -229,7 +233,7 @@ export class Precache { signal: this.aborter.signal, }), ).finally(() => { - this.originalsRunning = false; + this.originalsSweep = undefined; }); } diff --git a/src/metadata-backup.ts b/src/metadata-backup.ts index d5877bd..6021f87 100644 --- a/src/metadata-backup.ts +++ b/src/metadata-backup.ts @@ -4,7 +4,12 @@ import * as jpeg from "jpeg-js"; import exifReader from "exif-reader"; import type { Client } from "./client.js"; import type { Library, Photo } from "./library/index.js"; -import { fetchMLData } from "./mldata-fetch.js"; +import { sanitizeFileName } from "./filename.js"; +import { + fetchMLDataBatch, + MLDATA_BATCH_SIZE, + type MLData, +} from "./mldata-fetch.js"; import type { EnteFile } from "./model/types.js"; export type ProgressCallback = (message: string) => void; @@ -14,88 +19,110 @@ export interface MetadataBackupOptions { onProgress?: ProgressCallback; } -const sanitizePath = (name: string): string => - name.replace(/[/\\:*?"<>|]/g, "_").replace(/^\.+/, "_"); - -// Extract the raw EXIF APP1 segment from JPEG bytes. Returns the EXIF -// data buffer (starting after the APP1 length field, at the "Exif\0\0" -// header) or undefined if no APP1 marker is found. -const extractExifFromJpeg = (buf: Uint8Array): Buffer | undefined => { - if (buf[0] !== 0xff || buf[1] !== 0xd8) return undefined; +// Find the raw EXIF APP1 segment in JPEG bytes. Returns `exif` (the segment +// data, starting at the "Exif\0\0" header) when there is one, nothing when the +// bytes are not a JPEG or carry no EXIF, and `error` when the segment layout is +// malformed. Each segment length is checked against the bytes that remain and +// each step moves forward by at least 4 bytes, so the scan ends on any input. +export const extractExifFromJpeg = ( + buf: Uint8Array, +): { exif?: Buffer; error?: string } => { + if (buf[0] !== 0xff || buf[1] !== 0xd8) return {}; let offset = 2; - while (offset < buf.length - 1) { - if (buf[offset] !== 0xff) return undefined; + while (offset < buf.length) { + if (offset + 2 > buf.length) + return { error: `truncated segment marker at byte ${offset}` }; + if (buf[offset] !== 0xff) + return { error: `no segment marker at byte ${offset}` }; const marker = buf[offset + 1]!; - if (marker === 0xda) break; // start of scan, no more markers - if (offset + 3 >= buf.length) break; + if (marker === 0xda) return {}; // start of scan, no more markers + if (offset + 4 > buf.length) + return { error: `truncated segment length at byte ${offset}` }; const len = (buf[offset + 2]! << 8) | buf[offset + 3]!; + // The length counts its own two bytes, so anything under 2 is invalid. + if (len < 2) + return { + error: `segment length ${len} at byte ${offset} is too small`, + }; + if (offset + 2 + len > buf.length) + return { + error: `segment length ${len} at byte ${offset} runs past the end of the file`, + }; if (marker === 0xe1) { - // APP1 — check for "Exif\0\0" header + // APP1 — check for "Exif\0\0" header. A length under 8 cannot hold + // the six-byte header, so the segment is not EXIF; below 6 the + // bytes compared would also lie past the segment. if ( + len >= 8 && buf[offset + 4] === 0x45 && buf[offset + 5] === 0x78 && buf[offset + 6] === 0x69 && buf[offset + 7] === 0x66 ) { - return Buffer.from( - buf.buffer, - buf.byteOffset + offset + 4, - len - 2, - ); + return { + exif: Buffer.from( + buf.buffer, + buf.byteOffset + offset + 4, + len - 2, + ), + }; } } offset += 2 + len; } - return undefined; + return { error: "file ends before the image data" }; }; -const extractImageMetadata = ( +// Extract dimensions, EXIF and XMP from a file's bytes. When the EXIF segment +// is malformed or cannot be parsed, the record carries the reason in +// `exifError`. +export const extractImageMetadata = ( fileBytes: Uint8Array, ): Record<string, unknown> | undefined => { + const result: Record<string, unknown> = {}; + + // Try to get dimensions from JPEG decode try { - const result: Record<string, unknown> = {}; - - // Try to get dimensions from JPEG decode - try { - const decoded = jpeg.decode(fileBytes, { - useTArray: true, - formatAsRGBA: false, - }); - result.format = "jpeg"; - result.width = decoded.width; - result.height = decoded.height; - } catch { - // Not a JPEG or corrupt; still try EXIF extraction - } - - const exifBuf = extractExifFromJpeg(fileBytes); - if (exifBuf) { - try { - result.exif = exifReader(exifBuf); - } catch { - result.exifRaw = exifBuf.toString("base64"); - } - } - - // Extract XMP (look for "http://ns.adobe.com/xap" in the bytes) - const xmpStart = Buffer.from(fileBytes).indexOf("<?xpacket begin"); - if (xmpStart !== -1) { - const xmpEnd = Buffer.from(fileBytes).indexOf( - "<?xpacket end", - xmpStart, - ); - if (xmpEnd !== -1) { - const end = Buffer.from(fileBytes).indexOf("?>", xmpEnd); - result.xmp = Buffer.from(fileBytes) - .subarray(xmpStart, end !== -1 ? end + 2 : xmpEnd + 50) - .toString("utf-8"); - } - } - - return Object.keys(result).length > 0 ? result : undefined; + const decoded = jpeg.decode(fileBytes, { + useTArray: true, + formatAsRGBA: false, + }); + result.format = "jpeg"; + result.width = decoded.width; + result.height = decoded.height; } catch { - return undefined; + // Not every original is a JPEG (PNG, HEIC, video), so a failed decode + // is expected and only means no dimensions; a malformed JPEG is still + // reported below through `exifError`. } + + const { exif, error } = extractExifFromJpeg(fileBytes); + if (error) result.exifError = error; + if (exif) { + try { + result.exif = exifReader(exif); + } catch (err) { + result.exifRaw = exif.toString("base64"); + result.exifError = err instanceof Error ? err.message : String(err); + } + } + + // Extract XMP (look for "http://ns.adobe.com/xap" in the bytes) + const xmpStart = Buffer.from(fileBytes).indexOf("<?xpacket begin"); + if (xmpStart !== -1) { + const xmpEnd = Buffer.from(fileBytes).indexOf( + "<?xpacket end", + xmpStart, + ); + if (xmpEnd !== -1) { + const end = Buffer.from(fileBytes).indexOf("?>", xmpEnd); + result.xmp = Buffer.from(fileBytes) + .subarray(xmpStart, end !== -1 ? end + 2 : xmpEnd + 50) + .toString("utf-8"); + } + } + + return Object.keys(result).length > 0 ? result : undefined; }; // Read a file's original bytes through the library's content cache and extract @@ -105,26 +132,23 @@ const extractImageMetadata = ( const extractExif = async ( photo: Photo, ): Promise<Record<string, unknown> | undefined> => { - try { - const { path } = await photo.original(); - const fileBytes = new Uint8Array(readFileSync(path)); - return extractImageMetadata(fileBytes); - } catch { - return undefined; - } + const { path } = await photo.original(); + const fileBytes = new Uint8Array(readFileSync(path)); + return extractImageMetadata(fileBytes); }; // Dump every decrypted metadata layer the account holds into a directory tree // of plain JSON: account, per-collection, and per-file records including the // private and public magic metadata and (by default) the ML data. Collections -// and files are enumerated from the library's cache rather than a fresh server -// scan; the ML fetch and EXIF extraction are unchanged. +// and files are enumerated from the library's cache, which the caller refreshes +// first. Returns how many ML data requests failed; their files are still +// written, with `mlDataError` in place of `mlData`. export const runMetadataBackup = async ( lib: Library, client: Client, outDir: string, opts?: MetadataBackupOptions, -): Promise<void> => { +): Promise<{ failedMLBatches: number }> => { const log = opts?.onProgress ?? (() => {}); const wantExif = opts?.exif ?? false; @@ -151,7 +175,7 @@ export const runMetadataBackup = async ( const col = lib.getCollection(album.collectionID); if (!col) continue; - const dirName = `${col.id}-${sanitizePath(col.name || "unnamed")}`; + const dirName = `${col.id}-${sanitizeFileName(col.name, "unnamed")}`; const colDir = join(outDir, "collections", dirName); mkdirSync(colDir, { recursive: true }); @@ -189,12 +213,31 @@ export const runMetadataBackup = async ( } } + // One failed request (retries exhausted) must not end the dump: its files + // get the reason in `mlDataError` and the other batches go on. log("Fetching ML data (face detections, CLIP embeddings)..."); - const mlDataMap = await fetchMLData( - client.getApiClient(), - [...fileKeys.keys()], - fileKeys, - ); + const mlDataMap = new Map<number, MLData>(); + const mlDataErrors = new Map<number, string>(); + let failedMLBatches = 0; + const fileIDs = [...fileKeys.keys()]; + for (let i = 0; i < fileIDs.length; i += MLDATA_BATCH_SIZE) { + const batch = fileIDs.slice(i, i + MLDATA_BATCH_SIZE); + try { + const result = await fetchMLDataBatch( + client.getApiClient(), + batch, + fileKeys, + ); + for (const [id, payload] of result) mlDataMap.set(id, payload); + } catch (err) { + const reason = err instanceof Error ? err.message : String(err); + failedMLBatches++; + log( + `ML data request for ${batch.length} file(s) failed: ${reason}`, + ); + for (const id of batch) mlDataErrors.set(id, reason); + } + } log(`Got ML data for ${mlDataMap.size} file(s)`); const writtenFileIDs = new Set<number>(); @@ -214,11 +257,18 @@ export const runMetadataBackup = async ( const ml = mlDataMap.get(file.id); if (ml) fileMeta.mlData = ml; + const mlError = mlDataErrors.get(file.id); + if (mlError) fileMeta.mlDataError = mlError; if (wantExif && !writtenFileIDs.has(file.id)) { log(`[${file.metadata.title}] Extracting EXIF...`); - const exifData = await extractExif(photo); - if (exifData) fileMeta.imageMetadata = exifData; + try { + const exifData = await extractExif(photo); + if (exifData) fileMeta.imageMetadata = exifData; + } catch (err) { + fileMeta.imageMetadataError = + err instanceof Error ? err.message : String(err); + } } writtenFileIDs.add(file.id); @@ -229,4 +279,5 @@ export const runMetadataBackup = async ( } log("Metadata backup complete."); + return { failedMLBatches }; }; diff --git a/src/mldata-fetch.ts b/src/mldata-fetch.ts index d1b58b0..e5a254d 100644 --- a/src/mldata-fetch.ts +++ b/src/mldata-fetch.ts @@ -5,9 +5,8 @@ // comes back encrypted under the file's own key and gzipped; decrypting and // gunzipping yields the JSON payload // `{ face: { faces: [...] }, clip: { embedding } }`. Ente caps a request at 200 -// ids, so `fetchMLData` batches for callers that want many at once while -// `fetchMLDataBatch` is the single-request unit the library submits to its -// request pool. +// ids, so callers that want many at once split them into batches of +// `MLDATA_BATCH_SIZE` and call `fetchMLDataBatch` once per batch. import { gunzipSync } from "node:zlib"; @@ -69,25 +68,3 @@ export const fetchMLDataBatch = async ( } return result; }; - -// Fetch ML data for arbitrarily many ids, batching at `MLDATA_BATCH_SIZE`. Used -// by the one-shot metadata backup; the library fetches through its request pool -// with `fetchMLDataBatch` instead. -export const fetchMLData = async ( - api: ApiClient, - fileIDs: number[], - fileKeys: Map<number, Uint8Array>, -): Promise<Map<number, MLData>> => { - const result = new Map<number, MLData>(); - for (let i = 0; i < fileIDs.length; i += MLDATA_BATCH_SIZE) { - const batch = fileIDs.slice(i, i + MLDATA_BATCH_SIZE); - for (const [id, payload] of await fetchMLDataBatch( - api, - batch, - fileKeys, - )) { - result.set(id, payload); - } - } - return result; -}; diff --git a/src/model/decrypt.ts b/src/model/decrypt.ts index 253e3f0..9137479 100644 --- a/src/model/decrypt.ts +++ b/src/model/decrypt.ts @@ -34,6 +34,28 @@ const FILE_TYPE_MAP: Record<number, FileType> = { const parseFileType = (n: number): FileType => FILE_TYPE_MAP[n] ?? "unknown"; +// The hash the uploading client recorded for the original's bytes, read the +// way the upstream client's `metadataHash` reads it: `hash` if present, +// otherwise, for a live photo from an older client that wrote the two parts +// separately, `<imageHash>:<videoHash>`. A field that is not a non-empty +// string counts as absent, and a file with no hash at all is normal. +const expectedHash = (json: Record<string, unknown>): string | undefined => { + const text = (v: unknown): string | undefined => + typeof v === "string" && v !== "" ? v : undefined; + const hash = text(json.hash); + if (hash !== undefined) return hash; + const imageHash = text(json.imageHash); + const videoHash = text(json.videoHash); + if ( + json.fileType === 2 && + imageHash !== undefined && + videoHash !== undefined + ) { + return `${imageHash}:${videoHash}`; + } + return undefined; +}; + export const decryptCollection = ( raw: RawCollection, keys: KeyMaterial, @@ -98,15 +120,24 @@ export const decryptFile = ( key, ); const metadataJSON = JSON.parse(new TextDecoder().decode(metadataBytes)); + if ( + typeof metadataJSON !== "object" || + metadataJSON === null || + Array.isArray(metadataJSON) + ) { + throw new Error(`file ${raw.id}: metadata is not a JSON object`); + } const metadata: FileMetadata = { - title: metadataJSON.title ?? "", + // The server controls this JSON: a title that is missing or not a + // string becomes "", never an arbitrary value. + title: typeof metadataJSON.title === "string" ? metadataJSON.title : "", fileType: parseFileType(metadataJSON.fileType ?? -1), creationTime: metadataJSON.creationTime ?? 0, modificationTime: metadataJSON.modificationTime ?? 0, latitude: metadataJSON.latitude, longitude: metadataJSON.longitude, - hash: metadataJSON.hash, + hash: expectedHash(metadataJSON), }; const magicMetadata = decryptMagicMetadata(raw.magicMetadata, key); diff --git a/src/model/types.ts b/src/model/types.ts index 65cc845..516407a 100644 --- a/src/model/types.ts +++ b/src/model/types.ts @@ -29,6 +29,9 @@ export interface FileMetadata { modificationTime: Microseconds; latitude?: number; longitude?: number; + // The content hash the uploader recorded (see `expectedHash` in + // decrypt.ts); `downloadFile` refuses an original that does not match it. + // Absent for files from very old clients. hash?: string; } diff --git a/src/retry.ts b/src/retry.ts index 4d5eb69..1198a79 100644 --- a/src/retry.ts +++ b/src/retry.ts @@ -88,18 +88,21 @@ const MAX_CAUSE_DEPTH = 8; // errno on the error it throws — it hangs the underlying socket error off // `cause`, sometimes more than one level down — so a classifier that only read // the top-level error would see a bare `Error` and call every dropped -// connection permanent. -const causeCodes = (err: unknown): string[] => { +// connection permanent. `complete` is false when the walk stopped at the +// depth limit with more of the chain still below it. +const causeCodes = (err: unknown): { codes: string[]; complete: boolean } => { const codes: string[] = []; let current: unknown = err; for (let depth = 0; depth < MAX_CAUSE_DEPTH; depth++) { - if (current === null || typeof current !== "object") break; + if (current === null || typeof current !== "object") { + return { codes, complete: true }; + } const { code, cause } = current as { code?: unknown; cause?: unknown }; if (typeof code === "string") codes.push(code); - if (cause === current) break; + if (cause === current) return { codes, complete: true }; current = cause; } - return codes; + return { codes, complete: current === null || typeof current !== "object" }; }; const isAbort = (err: unknown): boolean => { @@ -145,15 +148,15 @@ export const isRetryable = (err: unknown): boolean => { // have succeeded; the cost of the imprecision is bounded by the attempt // count. if (err instanceof TypeError) return true; - return causeCodes(err).some((code) => TRANSPORT_CODES.has(code)); + return causeCodes(err).codes.some((code) => TRANSPORT_CODES.has(code)); }; // Could the first attempt already have taken effect on the server? // // `isRetryable` is the wrong question for a request that changes state. -// quak's non-idempotent calls are `/users/srp/create-session`, -// `/users/two-factor/verify` — which consumes one of a small number of 2FA -// attempts — and `/files/thumbnail`. They are replayed only on the failures in +// `postJSON` and `putJSON` use this for every `POST` and `PUT` listed in the +// README under "Endpoints used"; verifying a second factor, for one, consumes +// one of a small number of attempts. They are replayed only on the failures in // `CONNECT_CODES`, which establish that no TCP connection to the server ever // existed: there was no address to connect to, or the peer refused the // connection outright. A request byte cannot have been transmitted, so the @@ -162,9 +165,19 @@ export const isRetryable = (err: unknown): boolean => { // Everything else is ambiguous. A 5xx proves the server did process the // request. A reset or a broken pipe can arrive after it was fully sent and // acted on. A routing errno can be delivered on an established socket. A -// deadline says nothing at all about the server's state. -export const isSafeToReplay = (err: unknown): boolean => - isRetryable(err) && causeCodes(err).some((code) => CONNECT_CODES.has(code)); +// deadline says nothing at all about the server's state. So every errno in the +// cause chain must be a connect errno: one other errno anywhere in the chain +// is doubt, and doubt is not replayed. A chain longer than the walk is doubt +// too: the links below the limit were never read. +export const isSafeToReplay = (err: unknown): boolean => { + const { codes, complete } = causeCodes(err); + return ( + isRetryable(err) && + complete && + codes.length > 0 && + codes.every((code) => CONNECT_CODES.has(code)) + ); +}; export interface WithRetryOptions extends RetryOptions { isRetryable?: (err: unknown) => boolean; diff --git a/src/thumbnails.ts b/src/thumbnails.ts index 4183b1e..bb51f1e 100644 --- a/src/thumbnails.ts +++ b/src/thumbnails.ts @@ -7,8 +7,21 @@ import { ApiError } from "./api/client.js"; import { encryptBlob, toBase64 } from "./crypto/index.js"; import type { EnteFile } from "./model/types.js"; -const THUMB_MAX_DIMENSION = 720; -const THUMB_JPEG_QUALITY = 50; +// The server refuses a thumbnail larger than the one it already records for the +// file (`thumbnail.size`, the encrypted size), so these encodings are tried +// from largest to smallest and the first that fits is uploaded. +const THUMB_ENCODINGS = [ + { maxDimension: 720, quality: 50 }, + { maxDimension: 720, quality: 30 }, + { maxDimension: 480, quality: 30 }, + { maxDimension: 320, quality: 20 }, + { maxDimension: 160, quality: 20 }, +]; + +// The server accepts a new thumbnail only from the file's owner, so files other +// people own in albums shared with this account are never checked or repaired. +const NOT_OWNED_REASON = + "owned by another account (only the owner can replace its thumbnail)"; export interface MissingThumbnailInfo { fileID: number; @@ -19,11 +32,12 @@ export interface MissingThumbnailInfo { // Three outcomes, not two. "fixed": a thumbnail was generated and uploaded. // "failed": something went wrong (download, encode, upload) and the file still -// has no thumbnail. "skipped": the file is a format this helper cannot -// regenerate — a video, or an image that is not a baseline JPEG. Skipped is a -// deliberate, expected outcome, not an error (issue #17): the repair path is -// JPEG-only because `jpeg-js` is, and a PNG or HEIC is left for a format-aware -// tool rather than reported as a failure. +// has no thumbnail. "skipped": the server would refuse any thumbnail for the +// file or this helper cannot regenerate it — a file another account owns, a +// recorded thumbnail size nothing fits within, a video, or an image that is +// not a baseline JPEG. Skipped is a deliberate, expected outcome, not an error +// (issue #17): the repair path is JPEG-only because `jpeg-js` is, and a PNG or +// HEIC is left for a format-aware tool rather than reported as a failure. export type ThumbnailFixStatus = "fixed" | "skipped" | "failed"; export interface ThumbnailFixResult { @@ -45,6 +59,7 @@ export type ProgressCallback = (message: string) => void; // exists, so it is logged and the file is left unreported. That distinction is // what stops `fix-missing-thumbnails` from regenerating and uploading over // thumbnails that were fine all along while the CDN was briefly returning 500s. +// Files another account owns are logged as skipped and not checked. export const listMissingThumbnails = async ( lib: Library, client: Client, @@ -52,6 +67,7 @@ export const listMissingThumbnails = async ( ): Promise<MissingThumbnailInfo[]> => { const log = onProgress ?? (() => {}); const api = client.getApiClient(); + const { userID } = client.whoami(); const missing: MissingThumbnailInfo[] = []; const seen = new Set<number>(); @@ -60,6 +76,13 @@ export const listMissingThumbnails = async ( for (const photo of album.photos.list()) { if (seen.has(photo.fileID)) continue; seen.add(photo.fileID); + const file = lib.getFile(album.collectionID, photo.fileID); + if (file && file.ownerID !== userID) { + log( + `[${album.name}] Skipping ${photo.title}: ${NOT_OWNED_REASON}`, + ); + continue; + } try { const stream = await api.getThumbnailStream(photo.fileID); const reader = stream.getReader(); @@ -135,17 +158,13 @@ const resizeRGBA = ( return dst; }; -const generateThumbnail = (fileBytes: Uint8Array): Uint8Array => { - const decoded = jpeg.decode(fileBytes, { - useTArray: true, - formatAsRGBA: true, - }); +const generateThumbnail = ( + decoded: { data: Uint8Array; width: number; height: number }, + maxDimension: number, + quality: number, +): Uint8Array => { const { width: srcW, height: srcH } = decoded; - const scale = Math.min( - THUMB_MAX_DIMENSION / srcW, - THUMB_MAX_DIMENSION / srcH, - 1, - ); + const scale = Math.min(maxDimension / srcW, maxDimension / srcH, 1); const dstW = Math.round(srcW * scale); const dstH = Math.round(srcH * scale); @@ -158,7 +177,7 @@ const generateThumbnail = (fileBytes: Uint8Array): Uint8Array => { const encoded = jpeg.encode( { data: pixels, width: dstW, height: dstH }, - THUMB_JPEG_QUALITY, + quality, ); return new Uint8Array(encoded.data); }; @@ -171,14 +190,35 @@ const generateThumbnail = (fileBytes: Uint8Array): Uint8Array => { const isJpeg = (bytes: Uint8Array): boolean => bytes.length >= 2 && bytes[0] === 0xff && bytes[1] === 0xd8; -// The reason a file cannot have a JPEG thumbnail regenerated for it from its -// metadata alone, before any bytes are fetched, or undefined when it might. A -// non-image (video, live photo) is unsupported outright; a still image still -// has to be checked against its actual bytes once downloaded. -const unsupportedByType = (file: EnteFile): string | undefined => { +// The reason a file cannot have a JPEG thumbnail regenerated for it, known from +// its record alone before any bytes are fetched, or undefined when it might. A +// still image still has to be checked against its actual bytes once +// downloaded. +const reasonToSkip = (file: EnteFile, userID: number): string | undefined => { + if (file.ownerID !== userID) { + return NOT_OWNED_REASON; + } if (file.metadata.fileType !== "image") { return `unsupported file type: ${file.metadata.fileType} (only JPEG images can be regenerated)`; } + if (!file.thumbnail.size) { + return `recorded thumbnail size is ${file.thumbnail.size ?? "unknown"} (the server refuses a thumbnail larger than the one it records)`; + } + return undefined; +}; + +// Encrypt the largest encoding of the decoded image whose ciphertext is no +// larger than `maxSize`, or return undefined when even the smallest is larger. +const encryptThumbnailWithin = ( + decoded: { data: Uint8Array; width: number; height: number }, + key: Uint8Array, + maxSize: number, +): { header: Uint8Array; ciphertext: Uint8Array } | undefined => { + for (const { maxDimension, quality } of THUMB_ENCODINGS) { + const thumbJpeg = generateThumbnail(decoded, maxDimension, quality); + const encrypted = encryptBlob(thumbJpeg, key); + if (encrypted.ciphertext.length <= maxSize) return encrypted; + } return undefined; }; @@ -197,6 +237,7 @@ export const fixMissingThumbnails = async ( const log = onProgress ?? (() => {}); const results: ThumbnailFixResult[] = []; const api = client.getApiClient(); + const { userID } = client.whoami(); // Resolve each requested fileID to its file record and owning album by // enumerating the library, each file taken from the first album that holds @@ -237,18 +278,19 @@ export const fixMissingThumbnails = async ( const { file, collectionName } = entry; const title = file.metadata.title; - const typeReason = unsupportedByType(file); - if (typeReason) { - log(`[${collectionName}] Skipping ${title}: ${typeReason}`); + const skipReason = reasonToSkip(file, userID); + if (skipReason) { + log(`[${collectionName}] Skipping ${title}: ${skipReason}`); results.push({ fileID, title, collection: collectionName, status: "skipped", - reason: typeReason, + reason: skipReason, }); continue; } + const maxSize = file.thumbnail.size!; try { const photo = lib.photos.byID({ fileID }); @@ -277,12 +319,28 @@ export const fixMissingThumbnails = async ( } log(`[${collectionName}] Generating thumbnail for ${title}...`); - const thumbJpeg = generateThumbnail(fileBytes); + const decoded = jpeg.decode(fileBytes, { + useTArray: true, + formatAsRGBA: true, + }); + const fitting = encryptThumbnailWithin(decoded, file.key, maxSize); + if (!fitting) { + const reason = `no thumbnail encoding fits the recorded thumbnail size of ${maxSize} bytes`; + log(`[${collectionName}] Skipping ${title}: ${reason}`); + results.push({ + fileID, + title, + collection: collectionName, + status: "skipped", + reason, + }); + continue; + } + const { header, ciphertext } = fitting; log( - `[${collectionName}] Encrypting and uploading thumbnail (${thumbJpeg.length} bytes)...`, + `[${collectionName}] Uploading thumbnail (${ciphertext.length} bytes)...`, ); - const { header, ciphertext } = encryptBlob(thumbJpeg, file.key); const md5 = createHash("md5").update(ciphertext).digest("base64"); const { objectKey, url } = await api.getUploadURL( ciphertext.length, diff --git a/test/api/client.test.ts b/test/api/client.test.ts index 2faf019..68ab7cf 100644 --- a/test/api/client.test.ts +++ b/test/api/client.test.ts @@ -38,14 +38,18 @@ * the network. The fake records every call for assertion. */ -import { describe, expect, it } from "vitest"; +import { describe, expect, it, vi } from "vitest"; import { ApiClient, ApiError, DEFAULT_DOWNLOAD_TIMEOUT_MS, DEFAULT_REQUEST_TIMEOUT_MS, } from "../../src/api/client.js"; -import type { RetryOptions } from "../../src/retry.js"; +import { + isRetryable, + isSafeToReplay, + type RetryOptions, +} from "../../src/retry.js"; // --------------------------------------------------------------------------- // Test helpers @@ -385,6 +389,103 @@ describe("ApiClient custom origins", () => { }); }); +describe("ApiClient request URLs", () => { + it("accepts a path with or without a leading slash", async () => { + const { fetch, calls } = recordingFetch( + jsonResponse({}), + jsonResponse({}), + jsonResponse({}), + ); + const client = new ApiClient({ fetch }); + await client.getJSON("health"); + await client.postJSON("users/ott", {}); + await client.putJSON("/files/thumbnail", {}); + + expect(calls.map((c) => c.url)).toEqual([ + "https://api.ente.io/health", + "https://api.ente.io/users/ott", + "https://api.ente.io/files/thumbnail", + ]); + }); + + it("accepts an apiOrigin with a trailing slash", async () => { + const { fetch, calls } = recordingFetch(jsonResponse({})); + const client = new ApiClient({ + fetch, + apiOrigin: "https://my-ente.example.com/", + }); + await client.getJSON("/health"); + + expect(calls[0]!.url).toBe("https://my-ente.example.com/health"); + }); + + it("keeps a base path in a self-hosted apiOrigin for every request", async () => { + const body = new Uint8Array([1]); + const { fetch, calls } = recordingFetch( + jsonResponse({}), + jsonResponse({}), + jsonResponse({}), + streamResponse(body), + streamResponse(body), + ); + const client = new ApiClient({ + fetch, + apiOrigin: "https://example.com/ente/", + }); + await client.getJSON("/collections/v2", { sinceTime: 0 }); + await client.postJSON("/users/ott", {}); + await client.putJSON("/files/thumbnail", {}); + await client.getFileStream(99); + await client.getThumbnailStream(77); + + expect(calls.map((c) => c.url)).toEqual([ + "https://example.com/ente/collections/v2?sinceTime=0", + "https://example.com/ente/users/ott", + "https://example.com/ente/files/thumbnail", + "https://example.com/ente/files/download/99", + "https://example.com/ente/files/preview/77", + ]); + }); + + it("percent-encodes query parameters and skips undefined ones", async () => { + const { fetch, calls } = recordingFetch(jsonResponse({})); + const client = new ApiClient({ fetch }); + await client.getJSON("/search", { + q: "a&b=c/d é", + limit: 5, + cursor: undefined, + }); + + const url = new URL(calls[0]!.url); + expect(url.pathname).toBe("/search"); + expect(url.search).toBe("?q=a%26b%3Dc%2Fd+%C3%A9&limit=5"); + expect(url.searchParams.get("q")).toBe("a&b=c/d é"); + }); + + it("rejects a path that carries its own query string", async () => { + const { fetch, calls } = recordingFetch(); + const client = new ApiClient({ fetch }); + + await expect(client.getJSON("/diff?sinceTime=0")).rejects.toThrow( + /must not contain "\?"/, + ); + await expect(client.postJSON("/users/ott?x=1", {})).rejects.toThrow( + /must not contain "\?"/, + ); + expect(calls).toHaveLength(0); + }); + + it("rejects a path that carries a fragment", async () => { + const { fetch, calls } = recordingFetch(); + const client = new ApiClient({ fetch }); + + await expect(client.getJSON("/diff#top")).rejects.toThrow( + /must not contain "\?" or "#"/, + ); + expect(calls).toHaveLength(0); + }); +}); + describe("ApiError", () => { it("throws ApiError on 4xx with status, code, requestID", async () => { const { fetch } = recordingFetch( @@ -639,17 +740,33 @@ describe("ApiClient retries", () => { expect(policy.baseDelayMs).toBe(7); expect(policy.maxDelayMs).toBe(11); }); + + it("does not let a caller change its settings through that policy", async () => { + const { fetch, calls } = scriptedFetch( + textResponse("boom", 500), + textResponse("boom", 500), + textResponse("boom", 500), + ); + const client = new ApiClient({ + fetch, + retry: { ...noWait, attempts: 2 }, + }); + + client.getRetryOptions().attempts = 3; + + expect(client.getRetryOptions().attempts).toBe(2); + await expect(client.getJSON("/x")).rejects.toBeInstanceOf(ApiError); + expect(calls).toHaveLength(2); + }); }); describe("ApiClient timeouts", () => { it("ships bounded default deadlines", () => { - // Asserted here so the README and the code cannot drift. Two numbers - // rather than one, because a deadline that is sane for a JSON call is - // nowhere near enough for a multi-gigabyte body, and a deadline long - // enough for that body would let a hung API call stall a backup for - // ten minutes. + // Asserted here so the README and the code cannot drift. The request + // deadline bounds a whole JSON call; the download deadline is an idle + // one, measured from the last byte that arrived. expect(DEFAULT_REQUEST_TIMEOUT_MS).toBe(30_000); - expect(DEFAULT_DOWNLOAD_TIMEOUT_MS).toBe(600_000); + expect(DEFAULT_DOWNLOAD_TIMEOUT_MS).toBe(60_000); }); it("attaches an abort signal to every request", async () => { @@ -693,6 +810,51 @@ describe("ApiClient timeouts", () => { expect(new Set(signals).size).toBe(3); }, 5000); + it("gives every retrying entry point a fresh deadline per attempt", async () => { + // A refused connection is retried by every entry point, the + // non-idempotent ones included. If the deadline were created once, + // outside the retry, both attempts would carry the same signal. + const entryPoints: [ + string, + () => Response, + (c: ApiClient) => unknown, + ][] = [ + ["getJSON", () => jsonResponse({}), (c) => c.getJSON("/a")], + ["postJSON", () => jsonResponse({}), (c) => c.postJSON("/b", {})], + ["putJSON", () => jsonResponse({}), (c) => c.putJSON("/c", {})], + [ + "putFile", + () => new Response(null, { status: 200 }), + (c) => c.putFile("https://s3.example/x", new Uint8Array([1])), + ], + [ + "getFileStream", + () => streamResponse(new Uint8Array([1])), + (c) => c.getFileStream(1), + ], + [ + "getThumbnailStream", + () => streamResponse(new Uint8Array([1])), + (c) => c.getThumbnailStream(1), + ], + ]; + for (const [name, success, call] of entryPoints) { + const { fetch, calls } = scriptedFetch( + errnoError("ECONNREFUSED", "connect ECONNREFUSED"), + success(), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + await call(client); + + expect(calls, name).toHaveLength(2); + const [first, second] = calls.map((c) => c.init?.signal); + expect(first, name).toBeInstanceOf(AbortSignal); + expect(second, name).toBeInstanceOf(AbortSignal); + expect(second, name).not.toBe(first); + } + }); + it("recovers when a later attempt answers in time", async () => { const { fetch, calls } = scriptedFetch(HANG, jsonResponse({ ok: 1 })); const client = new ApiClient({ @@ -718,25 +880,91 @@ describe("ApiClient timeouts", () => { // that never produces a chunk and never observes the signal, so the // only thing that can unblock the read is quak's own enforcement of // the deadline over the stream it hands out. - const stalling = new Response( - new ReadableStream<Uint8Array>({ - pull: () => new Promise<void>(() => {}), - }), - { status: 200 }, - ); - const { fetch } = scriptedFetch(stalling); - const client = new ApiClient({ - fetch, - downloadTimeoutMs: 20, - retry: { ...noWait, attempts: 1 }, - }); + // + // The clock is faked, so the test runs under the real default + // deadline and waits for nothing. + vi.useFakeTimers(); + try { + const stalling = new Response( + new ReadableStream<Uint8Array>({ + pull: () => new Promise<void>(() => {}), + }), + { status: 200 }, + ); + const { fetch } = scriptedFetch(stalling); + const client = new ApiClient({ + fetch, + retry: { ...noWait, attempts: 1 }, + }); - const stream = await client.getFileStream(42); - const err: unknown = await readAll(stream).catch((e: unknown) => e); + const stream = await client.getFileStream(42); + let settled = false; + const result = readAll(stream).then( + (n) => n, + (e: unknown) => e, + ); + void result.finally(() => { + settled = true; + }); - expect(err).toBeInstanceOf(Error); - expect((err as Error).name).toBe("TimeoutError"); - }, 5000); + await vi.advanceTimersByTimeAsync(DEFAULT_DOWNLOAD_TIMEOUT_MS - 1); + expect(settled).toBe(false); + await vi.advanceTimersByTimeAsync(1); + + const err = await result; + expect(err).toBeInstanceOf(Error); + expect((err as Error).name).toBe("TimeoutError"); + // Classified as every deadline is: retried by the idempotent + // downloads, never replayed for a POST or PUT. + expect(isRetryable(err)).toBe(true); + expect(isSafeToReplay(err)).toBe(false); + } finally { + vi.useRealTimers(); + } + }); + + it("does not abort a slow body that keeps making progress", async () => { + // A deadline over the whole transfer would cut off a large video on a + // slow link however steadily it was arriving. The deadline restarts + // with every chunk, so a body that sends one byte every 600 ms for + // well over the 1000 ms deadline completes. + vi.useFakeTimers(); + try { + let sent = 0; + const trickling = new Response( + new ReadableStream<Uint8Array>({ + async pull(controller) { + await new Promise((resolve) => + setTimeout(resolve, 600), + ); + if (sent === 10) { + controller.close(); + return; + } + controller.enqueue(new Uint8Array([sent++])); + }, + }), + { status: 200 }, + ); + const { fetch } = scriptedFetch(trickling); + const client = new ApiClient({ + fetch, + downloadTimeoutMs: 1000, + retry: { ...noWait, attempts: 1 }, + }); + + const stream = await client.getFileStream(42); + const result = readAll(stream).then( + (n) => n, + (e: unknown) => e, + ); + await vi.advanceTimersByTimeAsync(11 * 600); + + expect(await result).toBe(10); + } finally { + vi.useRealTimers(); + } + }); it("lets a body that arrives in time through untouched", async () => { // The counterpart to the previous test: enforcing the deadline over @@ -762,6 +990,52 @@ describe("ApiClient timeouts", () => { } expect(joined).toEqual(payload); }); + + it("leaves no timer pending after a download completes or fails", async () => { + vi.useFakeTimers(); + try { + const { fetch } = scriptedFetch( + streamResponse(new Uint8Array([1, 2, 3])), + textResponse("gone", 404), + ); + const client = new ApiClient({ + fetch, + retry: { ...noWait, attempts: 1 }, + }); + + expect(await readAll(await client.getFileStream(1))).toBe(3); + expect(vi.getTimerCount()).toBe(0); + + await expect(client.getFileStream(2)).rejects.toBeInstanceOf( + ApiError, + ); + expect(vi.getTimerCount()).toBe(0); + } finally { + vi.useRealTimers(); + } + }); + + it("never lets the download timer keep the process alive", async () => { + const spy = vi.spyOn(globalThis, "setTimeout"); + try { + const { fetch } = scriptedFetch( + streamResponse(new Uint8Array([1])), + ); + const client = new ApiClient({ + fetch, + downloadTimeoutMs: 12_345, + retry: noWait, + }); + + const stream = await client.getFileStream(1); + const i = spy.mock.calls.findIndex((call) => call[1] === 12_345); + const timer = spy.mock.results[i]!.value as NodeJS.Timeout; + expect(timer.hasRef()).toBe(false); + await stream.cancel(); + } finally { + spy.mockRestore(); + } + }); }); describe("ApiClient error typing", () => { @@ -820,9 +1094,8 @@ describe("ApiClient error typing", () => { describe("ApiClient non-idempotent requests", () => { /** - * `postJSON` and `putJSON` carry quak's only requests that change server - * state: `/users/srp/create-session`, `/users/two-factor/verify` — which - * consumes one of a small number of 2FA attempts — and `/files/thumbnail`. + * `postJSON` and `putJSON` carry quak's requests that can change server + * state; the README lists them under "Endpoints used". * * They are retried only on a failure that establishes no TCP connection to * the server ever existed — DNS produced no address, or the peer refused @@ -922,4 +1195,30 @@ describe("ApiClient non-idempotent requests", () => { await refusedClient.updateThumbnail(1, "key", "header"); expect(refused.calls).toHaveLength(2); }); + + it("does not follow or replay a redirect on POST or PUT", async () => { + // The origin has already received a request it answers with a + // redirect, so following it would let a refused connection to the + // redirect target pass for a request that never went out. + for (const send of [ + (c: ApiClient) => c.postJSON("/users/ott", {}), + (c: ApiClient) => c.putJSON("/files/thumbnail", {}), + ]) { + const { fetch, calls } = scriptedFetch( + new Response(null, { + status: 307, + headers: { location: "https://elsewhere.example/" }, + }), + jsonResponse({}), + ); + const client = new ApiClient({ fetch, retry: noWait }); + + const err: unknown = await send(client).catch((e: unknown) => e); + + expect(calls[0]?.init?.redirect).toBe("manual"); + expect(err).toBeInstanceOf(ApiError); + expect((err as ApiError).status).toBe(307); + expect(calls).toHaveLength(1); + } + }); }); diff --git a/test/cli/backup.test.ts b/test/cli/backup.test.ts index 7823730..7bcf3bd 100644 --- a/test/cli/backup.test.ts +++ b/test/cli/backup.test.ts @@ -36,20 +36,52 @@ import { lstatSync, mkdirSync, mkdtempSync, + readdirSync, readFileSync, readlinkSync, rmSync, + symlinkSync, writeFileSync, } from "node:fs"; +import { spawnSync } from "node:child_process"; import { join } from "node:path"; import { tmpdir } from "node:os"; -import { describe, it, expect, beforeEach, afterEach } from "vitest"; +import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; +import { runBackup, type BackupLibrary } from "../../src/backup.js"; import { Library } from "../../src/library/index.js"; import type { ContentSource } from "../../src/library/content.js"; import type { CollectionsPage, FilesPage } from "../../src/client.js"; import type { Collection, EnteFile } from "../../src/model/types.js"; +// `open` and `rename` are wrapped to record, in order, every fsync and rename, +// so a test can pin the sequence "fsync the temp file, rename, fsync the +// directory" that makes a copied original survive a power cut. `vi.hoisted` +// because `vi.mock` factories run before module-level constants exist. +const fsEvents = vi.hoisted(() => [] as string[]); + +vi.mock("node:fs/promises", async (importOriginal) => { + const actual = await importOriginal<typeof import("node:fs/promises")>(); + return { + ...actual, + open: async ( + ...args: Parameters<typeof actual.open> + ): Promise<Awaited<ReturnType<typeof actual.open>>> => { + const handle = await actual.open(...args); + const realSync = handle.sync.bind(handle); + handle.sync = async (): Promise<void> => { + fsEvents.push(`sync:${String(args[0])}`); + await realSync(); + }; + return handle; + }, + rename: async (from: string, to: string): Promise<void> => { + fsEvents.push(`rename:${to}`); + await actual.rename(from, to); + }, + }; +}); + const USER_ID = 42; // Decrypted-byte length each stub original writes, keyed by fileID. @@ -107,6 +139,29 @@ class MockClient { } } +// A server that names an album and a file so as to climb out of the backup +// directory. +class HostileClient extends MockClient { + override async collectionsSince(): Promise<CollectionsPage> { + const page = await super.collectionsSince(); + return { + ...page, + collections: page.collections.length + ? [collection(3, "../escape")] + : [], + }; + } + override async filesSince(args: { + collectionID: number; + }): Promise<FilesPage> { + const files = + args.collectionID === 3 + ? [file(300, 3, "../../.ssh/authorized_keys")] + : []; + return { files, deleted: [], cursor: 1 }; + } +} + // A content source that writes byte buffers of the expected length and can be // told to fail one fileID's original, to exercise per-file resilience. interface StubSource extends ContentSource { @@ -136,9 +191,12 @@ const stubSource = (): StubSource => { let root: string; -const openLibrary = (source: ContentSource): Promise<Library> => +const openLibrary = ( + source: ContentSource, + client: MockClient = new MockClient(), +): Promise<Library> => Library.open({ - client: new MockClient(), + client, cacheDirectory: join(root, "cache"), contentSource: source, refreshIntervalSeconds: 3600, @@ -243,6 +301,30 @@ describe("lib.backup", () => { lib.close(); }); + it("keeps server-supplied album and file names inside the backup", async () => { + const lib = await openLibrary(stubSource(), new HostileClient()); + const outDir = join(root, "backup"); + + const result = await lib.backup({ downloadDirectory: outDir }); + + expect(result.failed).toBe(0); + // The title has no usable extension, so the original is `.bin`. + expect(existsSync(join(outDir, "originals", "300.bin"))).toBe(true); + const link = join( + outDir, + "collections", + "__escape", + "__.._.ssh_authorized_keys", + ); + expect(lstatSync(link).isSymbolicLink()).toBe(true); + expect(existsSync(join(outDir, "collections", "__escape.json"))).toBe( + true, + ); + // Nothing landed beside or above the backup directory. + expect(readdirSync(root).sort()).toEqual(["backup", "cache"]); + lib.close(); + }); + it("is an idempotent no-op when every original is already present", async () => { const source = stubSource(); const lib = await openLibrary(source); @@ -479,4 +561,496 @@ describe("lib.backup", () => { expect(readLedger(outDir).files["101"]!.attempts).toBe(1); lib.close(); }); + + it("fetches each original once and writes it only into the backup", async () => { + // A backup of a 500 GB account must write 500 GB, not a copy in the + // cache as well: an original fetched for the backup goes straight + // into its originals/, and the cache records it there. + const source = stubSource(); + const lib = await openLibrary(source); + const outDir = join(root, "backup"); + + const result = await lib.backup({ downloadDirectory: outDir }); + + expect(result.downloaded).toBe(3); + expect(source.originalCalls).toBe(3); + expect(readdirSync(join(root, "cache", "originals"))).toEqual([]); + const stored = readdirSync(join(outDir, "originals")).filter( + (name) => !name.endsWith(".json"), + ); + expect(stored.sort()).toEqual(["100.jpg", "101.jpg", "200.png"]); + // The cache counts the backup's copy as present: reading the + // original afterwards fetches nothing and answers with that copy. + const read = await lib.photos.byID({ fileID: 100 })!.original(); + expect(read.path).toBe(join(outDir, "originals", "100.jpg")); + expect(source.originalCalls).toBe(3); + await lib.close(); + }); + + it("fsyncs an original copied from the cache before the rename and its directory after", async () => { + const lib = await openLibrary(stubSource()); + const outDir = join(root, "backup"); + const originals = join(outDir, "originals"); + const dest = join(originals, "100.jpg"); + // Only an original already in the cache is copied into the backup; + // one fetched for the backup is written there by the download writer. + await lib.photos.byID({ fileID: 100 })!.original(); + fsEvents.length = 0; + + await lib.backup({ downloadDirectory: outDir }); + + const at = fsEvents.indexOf(`rename:${dest}`); + expect(at).toBeGreaterThan(0); + expect(fsEvents[at - 1]).toMatch( + /^sync:.*\/\.quak-backup-100\.jpg-\d+-[0-9a-z]*\.tmp$/, + ); + expect(fsEvents[at + 1]).toBe(`sync:${originals}`); + lib.close(); + }); + + it("removes temp files left by a killed backup but not those of one still running", async () => { + const outDir = join(root, "backup"); + const originals = join(outDir, "originals"); + mkdirSync(originals, { recursive: true }); + // A child that has already exited: its process ID is not running. + const exitedPID = spawnSync(process.execPath, ["-e", ""]).pid; + const leftover = `.quak-backup-100.jpg-${exitedPID}-abc123.tmp`; + // This test's own process stands in for a backup running at the same + // time. + const inProgress = `.quak-backup-101.jpg-${process.pid}-def456.tmp`; + writeFileSync(join(originals, leftover), "partial"); + writeFileSync(join(originals, inProgress), "partial"); + const lib = await openLibrary(stubSource()); + + await lib.backup({ downloadDirectory: outDir }); + + const names = readdirSync(originals); + expect(names).not.toContain(leftover); + expect(names).toContain(inProgress); + lib.close(); + }); + + it("removes leftover temp files in thumbnails/ but not those of a backup still running", async () => { + const outDir = join(root, "backup"); + const thumbnails = join(outDir, "thumbnails"); + mkdirSync(thumbnails, { recursive: true }); + const exitedPID = spawnSync(process.execPath, ["-e", ""]).pid; + const leftover = `.quak-backup-100.jpg-${exitedPID}-abc123.tmp`; + const inProgress = `.quak-backup-101.jpg-${process.pid}-def456.tmp`; + writeFileSync(join(thumbnails, leftover), "partial"); + writeFileSync(join(thumbnails, inProgress), "partial"); + const lib = await openLibrary(stubSource()); + + await lib.backup({ + downloadDirectory: outDir, + includeThumbnails: true, + }); + + const names = readdirSync(thumbnails); + expect(names).not.toContain(leftover); + expect(names).toContain(inProgress); + lib.close(); + }); +}); + +// Every refresh fails, as with an expired session or no network. +class FailingClient extends MockClient { + override async collectionsSince(): Promise<CollectionsPage> { + throw new Error("HTTP 401 from server"); + } +} + +// Holds its refresh open until `release()` is called, then reports a third +// album, so a backup can be started while that refresh is still running. +class HeldClient extends MockClient { + release!: () => void; + private held = new Promise<void>((resolve) => { + this.release = resolve; + }); + override async collectionsSince(): Promise<CollectionsPage> { + await this.held; + return { + collections: [collection(3, "Later")], + deleted: [], + cursor: 2, + }; + } + override async filesSince(args: { + collectionID: number; + }): Promise<FilesPage> { + if (args.collectionID !== 3) return super.filesSince(args); + return { files: [file(300, 3, "late.jpg")], deleted: [], cursor: 2 }; + } +} + +// Fill the library cache on disk, so the next open starts its refresh in the +// background instead of waiting for it. +const fillCache = async (): Promise<void> => { + const lib = await openLibrary(stubSource()); + await lib.close(); +}; + +describe("the refresh before a backup", () => { + it("waits for a refresh already running and backs up what it found", async () => { + await fillCache(); + const client = new HeldClient(); + const lib = await openLibrary(stubSource(), client); + const outDir = join(root, "backup"); + + const backup = lib.backup({ downloadDirectory: outDir }); + client.release(); + const result = await backup; + + expect(result.totalFiles).toBe(4); + expect(existsSync(join(outDir, "originals", "300.jpg"))).toBe(true); + await lib.close(); + }); + + it("fails before any download when the refresh fails, leaving failures.json as it was", async () => { + await fillCache(); + const outDir = join(root, "backup"); + seedLedger(outDir, 100, "beach.jpg"); + const ledgerPath = join(outDir, "failures.json"); + const ledgerBefore = readFileSync(ledgerPath, "utf-8"); + const source = stubSource(); + const lib = await openLibrary(source, new FailingClient()); + + await expect(lib.backup({ downloadDirectory: outDir })).rejects.toThrow( + "HTTP 401 from server", + ); + + expect(source.originalCalls).toBe(0); + expect(readFileSync(ledgerPath, "utf-8")).toBe(ledgerBefore); + expect(existsSync(join(outDir, "originals"))).toBe(false); + await lib.close(); + }); + + it("fails when the refresh fails on an empty cache, instead of backing up nothing", async () => { + const source = stubSource(); + const lib = await openLibrary(source, new FailingClient()); + const outDir = join(root, "backup"); + + await expect(lib.backup({ downloadDirectory: outDir })).rejects.toThrow( + "HTTP 401 from server", + ); + + expect(source.originalCalls).toBe(0); + expect(existsSync(outDir)).toBe(false); + await lib.close(); + }); +}); + +// The album folders under collections/, driven through `runBackup` with a +// stand-in library whose albums a test changes between runs. +describe("backup album folders", () => { + interface Album { + collection: Collection; + files: EnteFile[]; + } + + const libraryOf = (albums: Album[]): BackupLibrary => ({ + refresh: async () => {}, + listCollections: () => albums.map((a) => a.collection), + listFiles: (id) => + albums.find((a) => a.collection.id === id)?.files ?? [], + original: async (fileID) => { + const path = join(root, `source-${fileID}`); + writeFileSync(path, `original ${fileID}`); + return { path }; + }, + thumbnail: async () => { + throw new Error("no thumbnails in this stand-in"); + }, + }); + + // Every entry under collections/, one level of directories deep, with each + // symlink's target. + const tree = (outDir: string): string[] => { + const lines: string[] = []; + const list = (dir: string, prefix: string): void => { + for (const name of readdirSync(dir).sort()) { + const path = join(dir, name); + const st = lstatSync(path); + if (st.isSymbolicLink()) { + lines.push(`${prefix}${name} -> ${readlinkSync(path)}`); + } else if (st.isDirectory() && prefix === "") { + lines.push(`${name}/`); + list(path, `${name}/`); + } else { + lines.push(`${prefix}${name}`); + } + } + }; + list(join(outDir, "collections"), ""); + return lines; + }; + + const albumID = (outDir: string, jsonName: string): number => + JSON.parse(readFileSync(join(outDir, "collections", jsonName), "utf-8")) + .id; + + it("gives every file and every album its own name when names repeat", async () => { + const outDir = join(root, "backup"); + const lib = libraryOf([ + { + collection: collection(10, "Trip"), + files: [ + file(1, 10, "IMG_0001.JPG"), + file(2, 10, "IMG_0001.JPG"), + file(4, 10, "img_0001.jpg"), + file(3, 10, "other.jpg"), + ], + }, + { + collection: collection(11, "Trip"), + files: [file(3, 11, "other.jpg")], + }, + ]); + + const result = await runBackup(lib, { downloadDirectory: outDir }); + + expect(result.failed).toBe(0); + expect(tree(outDir)).toEqual([ + "Trip (10)/", + "Trip (10)/IMG_0001 (1).JPG -> ../../originals/1.JPG", + "Trip (10)/IMG_0001 (2).JPG -> ../../originals/2.JPG", + "Trip (10)/img_0001 (4).jpg -> ../../originals/4.jpg", + "Trip (10)/other.jpg -> ../../originals/3.jpg", + "Trip (10).json", + "Trip (11)/", + "Trip (11)/other.jpg -> ../../originals/3.jpg", + "Trip (11).json", + ]); + expect(albumID(outDir, "Trip (10).json")).toBe(10); + expect(albumID(outDir, "Trip (11).json")).toBe(11); + }); + + it("keeps names unique when a name with an ID added is another entry's own name", async () => { + const outDir = join(root, "backup"); + const lib = libraryOf([ + { + collection: collection(10, "Trip"), + files: [ + file(5, 10, "IMG (6).JPG"), + file(6, 10, "IMG.JPG"), + file(7, 10, "IMG.JPG"), + ], + }, + { + collection: collection(11, "Trip"), + files: [file(8, 11, "a.jpg")], + }, + { + collection: collection(12, "Trip (11)"), + files: [file(9, 12, "b.jpg")], + }, + ]); + + const result = await runBackup(lib, { downloadDirectory: outDir }); + + expect(result.failed).toBe(0); + expect(tree(outDir)).toEqual([ + "Trip (10)/", + "Trip (10)/IMG (6) (5).JPG -> ../../originals/5.JPG", + "Trip (10)/IMG (6).JPG -> ../../originals/6.JPG", + "Trip (10)/IMG (7).JPG -> ../../originals/7.JPG", + "Trip (10).json", + "Trip (11)/", + "Trip (11)/a.jpg -> ../../originals/8.jpg", + "Trip (11) (12)/", + "Trip (11) (12)/b.jpg -> ../../originals/9.jpg", + "Trip (11) (12).json", + "Trip (11).json", + ]); + expect(albumID(outDir, "Trip (10).json")).toBe(10); + expect(albumID(outDir, "Trip (11).json")).toBe(11); + expect(albumID(outDir, "Trip (11) (12).json")).toBe(12); + }); + + it("changes nothing on a second run over an unchanged account", async () => { + const outDir = join(root, "backup"); + const lib = libraryOf([ + { + collection: collection(10, "Trip"), + files: [ + file(1, 10, "IMG_0001.JPG"), + file(2, 10, "IMG_0001.JPG"), + ], + }, + { + collection: collection(11, "Trip"), + files: [file(3, 11, "other.jpg")], + }, + ]); + + await runBackup(lib, { downloadDirectory: outDir }); + const before = tree(outDir); + const second = await runBackup(lib, { downloadDirectory: outDir }); + + expect(second.downloaded).toBe(0); + expect(second.failed).toBe(0); + expect(tree(outDir)).toEqual(before); + }); + + it("leaves the albums an onlyAlbumNames run skips as they were", async () => { + const outDir = join(root, "backup"); + // "trip" is skipped by the scoped run but its name clashes with the + // in-scope "Trip", so "Trip" must keep its ID suffix. + const lib = libraryOf([ + { + collection: collection(10, "Trip"), + files: [file(1, 10, "a.jpg")], + }, + { + collection: collection(11, "trip"), + files: [file(2, 11, "b.jpg")], + }, + { + collection: collection(12, "Work"), + files: [file(3, 12, "c.jpg")], + }, + ]); + const json = (name: string): string => + readFileSync(join(outDir, "collections", name), "utf-8"); + + await runBackup(lib, { downloadDirectory: outDir }); + const before = tree(outDir); + const skippedJSON = [json("trip (11).json"), json("Work.json")]; + const scoped = await runBackup(lib, { + downloadDirectory: outDir, + onlyAlbumNames: ["Trip"], + }); + + expect(scoped.failed).toBe(0); + expect(before).toEqual([ + "Trip (10)/", + "Trip (10)/a.jpg -> ../../originals/1.jpg", + "Trip (10).json", + "Work/", + "Work/c.jpg -> ../../originals/3.jpg", + "Work.json", + "trip (11)/", + "trip (11)/b.jpg -> ../../originals/2.jpg", + "trip (11).json", + ]); + expect(tree(outDir)).toEqual(before); + expect([json("trip (11).json"), json("Work.json")]).toEqual( + skippedJSON, + ); + }); + + it("removes links and album folders that are gone, and nothing the user added", async () => { + const outDir = join(root, "backup"); + const albums: Album[] = [ + { + collection: collection(10, "Trip"), + files: [ + file(1, 10, "IMG_0001.JPG"), + file(2, 10, "IMG_0001.JPG"), + file(3, 10, "other.jpg"), + ], + }, + { + collection: collection(12, "Work"), + files: [file(5, 12, "a.jpg")], + }, + { + collection: collection(13, "Old"), + files: [file(5, 13, "a.jpg")], + }, + ]; + const lib = libraryOf(albums); + await runBackup(lib, { downloadDirectory: outDir }); + + // What the user put in the tree: a note and a symlink of their own in + // an album, a note in an album about to be renamed, and a folder quak + // did not create. + const collectionsDir = join(outDir, "collections"); + writeFileSync(join(collectionsDir, "Trip", "notes.txt"), "mine"); + symlinkSync("../elsewhere", join(collectionsDir, "Trip", "mine")); + writeFileSync(join(collectionsDir, "Work", "keep.txt"), "mine"); + mkdirSync(join(collectionsDir, "Mine")); + writeFileSync(join(collectionsDir, "Mine", "keep.txt"), "mine"); + + // File 2 leaves Trip, Work is renamed Office, Old is deleted. + albums[0]!.files.splice(1, 1); + albums[1]!.collection = collection(12, "Office"); + albums.splice(2, 1); + const result = await runBackup(lib, { downloadDirectory: outDir }); + + expect(result.failed).toBe(0); + expect(tree(outDir)).toEqual([ + "Mine/", + "Mine/keep.txt", + "Office/", + "Office/a.jpg -> ../../originals/5.jpg", + "Office.json", + "Trip/", + "Trip/IMG_0001.JPG -> ../../originals/1.JPG", + "Trip/mine -> ../elsewhere", + "Trip/notes.txt", + "Trip/other.jpg -> ../../originals/3.jpg", + "Trip.json", + "Work/", + "Work/keep.txt", + "Work.json", + ]); + }); + + // One album backed up, then a folder the user made beside it holding a + // symlink into originals/, with `json` (if given) as its sibling JSON. + const backupWithUserFolder = async ( + json: string | undefined, + ): Promise<{ outDir: string; failed: number }> => { + const outDir = join(root, "backup"); + const lib = libraryOf([ + { + collection: collection(10, "Trip"), + files: [file(1, 10, "a.jpg")], + }, + ]); + await runBackup(lib, { downloadDirectory: outDir }); + const collectionsDir = join(outDir, "collections"); + mkdirSync(join(collectionsDir, "Mine")); + symlinkSync( + "../../originals/1.jpg", + join(collectionsDir, "Mine", "a.jpg"), + ); + if (json !== undefined) { + writeFileSync(join(collectionsDir, "Mine.json"), json); + } + const result = await runBackup(lib, { downloadDirectory: outDir }); + return { outDir, failed: result.failed }; + }; + + it("leaves a user folder with no JSON beside it as it was", async () => { + const { outDir, failed } = await backupWithUserFolder(undefined); + + expect(failed).toBe(0); + expect(tree(outDir)).toEqual([ + "Mine/", + "Mine/a.jpg -> ../../originals/1.jpg", + "Trip/", + "Trip/a.jpg -> ../../originals/1.jpg", + "Trip.json", + ]); + }); + + it("leaves a user folder whose JSON has no album ID as it was", async () => { + const json = '{"name":"Mine"}'; + const { outDir, failed } = await backupWithUserFolder(json); + + expect(failed).toBe(0); + expect(tree(outDir)).toEqual([ + "Mine/", + "Mine/a.jpg -> ../../originals/1.jpg", + "Mine.json", + "Trip/", + "Trip/a.jpg -> ../../originals/1.jpg", + "Trip.json", + ]); + expect( + readFileSync(join(outDir, "collections", "Mine.json"), "utf-8"), + ).toBe(json); + }); }); diff --git a/test/cli/commands.test.ts b/test/cli/commands.test.ts new file mode 100644 index 0000000..7585335 --- /dev/null +++ b/test/cli/commands.test.ts @@ -0,0 +1,732 @@ +/** + * Tests for the CLI commands (`src/cli-commands.ts`, issue #12). + * + * Each command is called directly with a context whose output streams collect + * text, whose session directory is a fresh temp directory, and whose session + * loader hands back a fake client. The fake serves two albums and three files + * from memory, writes stand-in bytes for originals and thumbnails, and makes no + * network calls. The helpers the commands call (`cli-read`, `cli-output`, + * backup, thumbnails) have their own tests; these check what each command + * prints and the exit code it returns. + */ + +import { + existsSync, + mkdtempSync, + readdirSync, + readFileSync, + rmSync, + statSync, + writeFileSync, +} from "node:fs"; +import { join } from "node:path"; +import { tmpdir } from "node:os"; +import { PassThrough } from "node:stream"; +import { + describe, + it, + expect, + vi, + beforeAll, + beforeEach, + afterEach, +} from "vitest"; + +import { + type CliContext, + saveSession, + loginCommand, + whoamiCommand, + logoutCommand, + collectionsCommand, + filesCommand, + getCommand, + getThumbCommand, + backupCommand, + backupMetadataCommand, + listMissingThumbnailsCommand, + fixMissingThumbnailsCommand, +} from "../../src/cli-commands.js"; +import { run } from "../../src/cli-run.js"; +import { loadSession } from "../../src/cli-session.js"; +import type { Client, ClientSnapshot, LoginOptions } from "../../src/client.js"; +import type { ContentSource } from "../../src/library/content.js"; +import type { Collection, EnteFile } from "../../src/model/types.js"; +import { init, toBase64 } from "../../src/crypto/index.js"; +import { defaultCacheDirectory } from "../../src/library/index.js"; + +const USER_ID = 42; + +const collection = ( + id: number, + name: string, + isShared = false, +): Collection => ({ + id, + ownerID: USER_ID, + key: new Uint8Array([id]), + name, + type: "album", + updationTime: 1, + isShared, +}); + +const file = (id: number, collectionID: number, title: string): EnteFile => ({ + id, + collectionID, + ownerID: USER_ID, + key: new Uint8Array([id & 0xff]), + metadata: { + title, + fileType: "image", + creationTime: 1000, + modificationTime: 1000, + }, + file: { decryptionHeader: "aGVhZGVy" }, + thumbnail: { decryptionHeader: "dGh1bWI=" }, + updationTime: 1, +}); + +const COLLECTIONS = [collection(1, "Vacation"), collection(2, "Work", true)]; + +const FILES: Record<number, EnteFile[]> = { + 1: [file(100, 1, "beach.jpg"), file(101, 1, "sunset.jpg")], + 2: [file(200, 2, "diagram.png")], +}; + +// An original is 7 bytes and a thumbnail 3. `failID` makes that file's +// original fail; `emptyThumbID` makes the server report that file's +// thumbnail as empty. `withNewFile` adds new.jpg (102) to Vacation, advancing +// the collection's updationTime as the server does, and `refreshError` makes +// listing collections fail with that message. +const fakeClient = ( + opts: { + failID?: number; + emptyThumbID?: number; + withNewFile?: boolean; + refreshError?: string; + } = {}, +) => { + const collections = opts.withNewFile + ? [{ ...COLLECTIONS[0], updationTime: 2 }, COLLECTIONS[1]] + : COLLECTIONS; + const files = opts.withNewFile + ? { + ...FILES, + 1: [...FILES[1], { ...file(102, 1, "new.jpg"), updationTime: 2 }], + } + : FILES; + const source: ContentSource = { + original: async ({ file: f, destination }) => { + if (f.id === opts.failID) throw new Error("HTTP 500 from server"); + writeFileSync(destination, Buffer.alloc(7, f.id & 0xff)); + return { bytesWritten: 7 }; + }, + thumbnail: async ({ file: f, destination }) => { + writeFileSync(destination, Buffer.alloc(3, f.id & 0xff)); + return { bytesWritten: 3 }; + }, + }; + const fake = { + whoami: () => ({ email: "cli@example.com", userID: USER_ID }), + collectionsSince: async () => { + if (opts.refreshError) throw new Error(opts.refreshError); + return { collections, deleted: [], cursor: 1 }; + }, + filesSince: async (args: { collectionID: number }) => ({ + files: files[args.collectionID] ?? [], + deleted: [], + cursor: 1, + }), + contentSource: () => source, + getApiClient: () => ({ + // The ML data request of `backup-metadata`: no file has any. + postJSON: async () => ({ data: [] }), + getThumbnailStream: async (fileID: number) => + new ReadableStream<Uint8Array>({ + start(controller) { + if (fileID !== opts.emptyThumbID) { + controller.enqueue(new Uint8Array(3)); + } + controller.close(); + }, + }), + }), + }; + // The commands only call the methods above. + return fake as unknown as Client; +}; + +// Collects everything written to it. +class Output { + text = ""; + write(text: string): void { + this.text += text; + } +} + +let root: string; +let stdout: Output; +let stderr: Output; + +const context = (client: Client | null = fakeClient()): CliContext => ({ + stdout, + stderr, + sessionDir: join(root, "session"), + cacheDir: join(root, "cache"), + loadSession: () => client, + login: async () => { + throw new Error("login not expected"); + }, + prompt: async () => { + throw new Error("prompt not expected"); + }, + promptSecret: async () => { + throw new Error("prompt not expected"); + }, +}); + +beforeAll(async () => { + await init(); +}); + +beforeEach(() => { + root = mkdtempSync(join(tmpdir(), "quak-cli-test-")); + stdout = new Output(); + stderr = new Output(); +}); + +afterEach(() => { + rmSync(root, { recursive: true, force: true }); +}); + +describe("session file", () => { + const snapshot: ClientSnapshot = { + email: "cli@example.com", + userID: USER_ID, + token: "token", + masterKey: "a", + secretKey: "b", + publicKey: "c", + }; + + it("is written with mode 0600 in a directory with mode 0700", () => { + const dir = join(root, "new", "session"); + saveSession(dir, snapshot); + expect(statSync(dir).mode & 0o777).toBe(0o700); + const path = join(dir, "session.json"); + expect(statSync(path).mode & 0o777).toBe(0o600); + expect(JSON.parse(readFileSync(path, "utf-8"))).toEqual(snapshot); + }); + + it("a missing session exits 1 with 'Not logged in'", async () => { + const ctx = { ...context(), loadSession }; + expect(await whoamiCommand(ctx)).toBe(1); + expect(stderr.text).toBe( + `Not logged in. Run "quak login" first.\n` + + `Session file: ${join(ctx.sessionDir, "session.json")}\n`, + ); + expect(stdout.text).toBe(""); + }); + + it("a corrupt session exits 1 and says it is corrupt", async () => { + const ctx = { ...context(), loadSession }; + saveSession(ctx.sessionDir, snapshot); + expect(await collectionsCommand(ctx, {})).toBe(1); + expect(stderr.text).toContain("is corrupt"); + expect(stderr.text).toContain( + `Run "quak logout" and then "quak login" to replace it.\n`, + ); + expect(stdout.text).toBe(""); + }); +}); + +// The login function is a fake that hands back a client whose snapshot is +// `snapshot`; each prompt is recorded and answered with "123456". +describe("login", () => { + const snapshot: ClientSnapshot = { + email: "cli@example.com", + userID: USER_ID, + token: "token", + masterKey: "a", + secretKey: "b", + publicKey: "c", + }; + + const loggedIn = { + whoami: () => ({ email: "cli@example.com", userID: USER_ID }), + toJSON: () => snapshot, + } as unknown as Client; + + let prompts: string[]; + + const loginContext = ( + login: (opts: LoginOptions) => Promise<Client>, + ): CliContext => ({ + ...context(), + login, + prompt: async (message) => { + prompts.push(message); + return "123456"; + }, + promptSecret: async (message) => { + prompts.push(message); + return "123456"; + }, + }); + + beforeEach(() => { + prompts = []; + vi.stubEnv("QUAK_EMAIL", "cli@example.com"); + vi.stubEnv("QUAK_PASSWORD", "hunter2"); + }); + + afterEach(() => { + vi.unstubAllEnvs(); + }); + + it("takes the email and password from the environment without a prompt", async () => { + const calls: LoginOptions[] = []; + const ctx = loginContext(async (opts) => { + calls.push(opts); + return loggedIn; + }); + expect(await loginCommand(ctx)).toBe(0); + + expect(prompts).toEqual([]); + expect(calls).toHaveLength(1); + expect(calls[0]!.email).toBe("cli@example.com"); + expect(calls[0]!.password).toBe("hunter2"); + const path = join(ctx.sessionDir, "session.json"); + expect(stderr.text).toBe( + "Authenticating...\n" + + `Logged in as cli@example.com (user ${USER_ID})\n` + + `Session saved to ${path}\n`, + ); + }); + + it("saves the session with mode 0600 in a directory with mode 0700", async () => { + const ctx = loginContext(async () => loggedIn); + expect(await loginCommand(ctx)).toBe(0); + + expect(statSync(ctx.sessionDir).mode & 0o777).toBe(0o700); + const path = join(ctx.sessionDir, "session.json"); + expect(statSync(path).mode & 0o777).toBe(0o600); + expect(JSON.parse(readFileSync(path, "utf-8"))).toEqual(snapshot); + }); + + it("asks for the TOTP code when the account needs one", async () => { + let code: string | undefined; + const ctx = loginContext(async (opts) => { + code = await opts.totp!(); + return loggedIn; + }); + expect(await loginCommand(ctx)).toBe(0); + + expect(prompts).toEqual(["TOTP code: "]); + expect(code).toBe("123456"); + }); + + it("a failed login exits 1, says why and writes no session", async () => { + const ctx = loginContext(async () => { + throw new Error("HTTP 401 from server"); + }); + expect(await loginCommand(ctx)).toBe(1); + + expect(stderr.text).toBe( + "Authenticating...\nLogin failed: HTTP 401 from server\n", + ); + expect(existsSync(join(ctx.sessionDir, "session.json"))).toBe(false); + }); +}); + +// These use a real client read from the session file, over a fake API that +// records each request and answers with `status`. +describe("logout", () => { + const snapshot: ClientSnapshot = { + email: "cli@example.com", + userID: USER_ID, + token: "saved-token", + masterKey: toBase64(new Uint8Array(32)), + secretKey: toBase64(new Uint8Array(32)), + publicKey: toBase64(new Uint8Array(32)), + }; + + const requests: Request[] = []; + + const logoutContext = (status: number): CliContext => ({ + ...context(), + loadSession: (path) => + loadSession(path, { + fetch: async (url, init) => { + requests.push(new Request(url, init)); + return new Response(JSON.stringify({}), { + status, + headers: { "content-type": "application/json" }, + }); + }, + }), + }); + + beforeEach(() => { + requests.length = 0; + }); + + it("ends the session on the server, then deletes the file", async () => { + const ctx = logoutContext(200); + saveSession(ctx.sessionDir, snapshot); + expect(await logoutCommand(ctx)).toBe(0); + + expect(requests).toHaveLength(1); + expect(requests[0]!.method).toBe("POST"); + expect(new URL(requests[0]!.url).pathname).toBe("/users/logout"); + expect(requests[0]!.headers.get("X-Auth-Token")).toBe("saved-token"); + expect(existsSync(join(ctx.sessionDir, "session.json"))).toBe(false); + expect(stderr.text).toBe( + "Session ended on the server.\n" + + "Session deleted.\n" + + `Cache directory ${ctx.cacheDir} still holds decrypted data; delete it to remove that data.\n`, + ); + }); + + it("still deletes the file when the server call fails, and says so", async () => { + const ctx = logoutContext(500); + saveSession(ctx.sessionDir, snapshot); + expect(await logoutCommand(ctx)).toBe(1); + + expect(requests).toHaveLength(1); + expect(existsSync(join(ctx.sessionDir, "session.json"))).toBe(false); + expect(stderr.text).toBe( + "Could not end the session on the server: HTTP 500\n" + + "Session deleted.\n" + + `Cache directory ${ctx.cacheDir} still holds decrypted data; delete it to remove that data.\n`, + ); + }); + + it("names the account's default cache directory without --cache-dir", async () => { + const ctx = { ...logoutContext(200), cacheDir: undefined }; + saveSession(ctx.sessionDir, snapshot); + expect(await logoutCommand(ctx)).toBe(0); + expect(stderr.text).toContain( + `Cache directory ${defaultCacheDirectory(USER_ID)} still holds decrypted data`, + ); + }); + + it("without a session says so, calls nothing and exits 0", async () => { + expect(await logoutCommand(logoutContext(200))).toBe(0); + expect(requests).toHaveLength(0); + expect(stderr.text).toBe("No session found.\n"); + }); +}); + +describe("whoami", () => { + it("prints the account as one line of JSON", async () => { + expect(await whoamiCommand(context())).toBe(0); + expect(stdout.text).toBe( + `{"email":"cli@example.com","userID":${USER_ID}}\n`, + ); + }); +}); + +describe("collections", () => { + it("prints one tab-separated line per album", async () => { + expect(await collectionsCommand(context(), {})).toBe(0); + expect(stdout.text).toBe( + "1\talbum\tVacation\n" + "2\talbum\tWork (shared)\n", + ); + }); + + it("prints a JSON array with --json", async () => { + expect(await collectionsCommand(context(), { json: true })).toBe(0); + expect(JSON.parse(stdout.text)).toEqual([ + { + id: 1, + name: "Vacation", + type: "album", + ownerID: USER_ID, + isShared: false, + updationTime: 1, + }, + { + id: 2, + name: "Work", + type: "album", + ownerID: USER_ID, + isShared: true, + updationTime: 1, + }, + ]); + }); +}); + +describe("files", () => { + it("prints one tab-separated line per file", async () => { + expect(await filesCommand(context(), { collection: "1" })).toBe(0); + expect(stdout.text).toBe( + "100\timage\tbeach.jpg\n" + "101\timage\tsunset.jpg\n", + ); + }); + + it("prints a JSON array with --json", async () => { + const code = await filesCommand(context(), { + collection: "2", + json: true, + }); + expect(code).toBe(0); + expect(JSON.parse(stdout.text)).toEqual([ + { + id: 200, + title: "diagram.png", + fileType: "image", + creationTime: 1000, + collectionID: 2, + }, + ]); + }); + + it("exits 1 for an unknown collection", async () => { + expect(await filesCommand(context(), { collection: "9" })).toBe(1); + expect(stderr.text).toBe("Collection 9 not found\n"); + }); + + it("exits 1 for a collection ID that is not a number", async () => { + expect(await filesCommand(context(), { collection: "abc" })).toBe(1); + expect(stderr.text).toBe("Invalid collection ID\n"); + }); +}); + +describe("get and get-thumb", () => { + it("get finds a file in any album without --collection", async () => { + const out = join(root, "diagram.png"); + expect(await getCommand(context(), "200", { out })).toBe(0); + expect(readFileSync(out)).toEqual(Buffer.alloc(7, 200)); + expect(stderr.text).toBe(`7 bytes -> ${out}\n`); + }); + + it("get-thumb finds a file in any album without --collection", async () => { + const out = join(root, "thumb.jpg"); + expect(await getThumbCommand(context(), "200", { out })).toBe(0); + expect(readFileSync(out)).toEqual(Buffer.alloc(3, 200)); + expect(stderr.text).toBe(`3 bytes -> ${out}\n`); + }); + + it("get exits 1 when no album has the file", async () => { + const out = join(root, "x"); + expect(await getCommand(context(), "999", { out })).toBe(1); + expect(stderr.text).toBe("File 999 not found\n"); + expect(existsSync(out)).toBe(false); + }); + + it("get-thumb exits 1 when no album has the file", async () => { + const out = join(root, "x"); + expect(await getThumbCommand(context(), "999", { out })).toBe(1); + expect(stderr.text).toBe("File 999 not found\n"); + expect(existsSync(out)).toBe(false); + }); + + it("both exit 1 for a file ID that is not a number", async () => { + expect(await getCommand(context(), "abc", {})).toBe(1); + expect(await getThumbCommand(context(), "abc", {})).toBe(1); + expect(stderr.text).toBe("Invalid file ID\nInvalid file ID\n"); + }); +}); + +describe("backup", () => { + it("exits 0 and prints a summary when every file is saved", async () => { + const dir = join(root, "backup"); + expect(await backupCommand(context(), dir, {})).toBe(0); + expect(stderr.text).toContain( + "\n--- Backup complete ---\n" + + " Total files: 3\n" + + " Downloaded: 3\n" + + " Skipped: 0\n" + + " Failed: 0\n", + ); + expect(stdout.text).toBe(""); + }); + + // The backup opens its library with the precache off: it fetches the + // originals it needs into the backup, and must not also fetch every + // thumbnail in the account, or keep originals, in the per-user cache. + it("leaves nothing in the cache's originals and thumbnails", async () => { + const dir = join(root, "backup"); + expect(await backupCommand(context(), dir, {})).toBe(0); + expect(readdirSync(join(root, "cache", "thumbnails"))).toEqual([]); + expect(readdirSync(join(root, "cache", "originals"))).toEqual([]); + }); + + it("exits 1 and lists the file when one download fails", async () => { + const ctx = context(fakeClient({ failID: 101 })); + expect(await backupCommand(ctx, join(root, "backup"), {})).toBe(1); + expect(stderr.text).toContain(" Failed: 1\n"); + expect(stderr.text).toContain( + "\nFailed files:\n" + + " [Vacation] sunset.jpg (id 101): HTTP 500 from server\n", + ); + }); + + it("prints the result as JSON with --json, still exiting 1 on a failure", async () => { + const ctx = context(fakeClient({ failID: 101 })); + const code = await backupCommand(ctx, join(root, "backup"), { + json: true, + }); + expect(code).toBe(1); + const result = JSON.parse(stdout.text); + expect(result).toMatchObject({ + totalFiles: 3, + downloaded: 2, + skipped: 0, + failed: 1, + }); + expect(result.errors[0].fileID).toBe(101); + expect(stderr.text).toBe("Starting backup...\n"); + }); + + it("exits 1 with the error on one line when the refresh fails", async () => { + const client = { + ...fakeClient(), + collectionsSince: async () => { + throw new Error("HTTP 401 from server"); + }, + } as unknown as Client; + const dir = join(root, "backup"); + // Through `run`, as `bin/quak.ts` does, which prints a thrown error. + const runStderr = new PassThrough(); + let runText = ""; + runStderr.on("data", (chunk: Buffer) => { + runText += chunk.toString(); + }); + const code = await new Promise<number>((resolve) => { + void run( + backupCommand(context(client), dir, {}), + new PassThrough(), + runStderr, + resolve, + ); + }); + expect(code).toBe(1); + expect(runText).toBe("quak: HTTP 401 from server\n"); + expect(stderr.text).toBe("Starting backup...\nRefreshing library...\n"); + expect(existsSync(join(dir, "originals"))).toBe(false); + }); +}); + +describe("helper list-missing-thumbnails", () => { + it("prints one line per file with an empty thumbnail", async () => { + const ctx = context(fakeClient({ emptyThumbID: 200 })); + expect(await listMissingThumbnailsCommand(ctx, {})).toBe(0); + expect(stdout.text).toBe( + "200\tdiagram.png\tWork\tempty thumbnail (0 bytes)\n", + ); + expect(stderr.text).toContain("\n1 file(s) with missing thumbnails:\n"); + }); + + it("says so when nothing is missing", async () => { + expect(await listMissingThumbnailsCommand(context(), {})).toBe(0); + expect(stdout.text).toBe(""); + expect(stderr.text).toContain("No missing thumbnails found.\n"); + }); + + it("prints a JSON array with --json and no progress", async () => { + const ctx = context(fakeClient({ emptyThumbID: 200 })); + expect(await listMissingThumbnailsCommand(ctx, { json: true })).toBe(0); + expect(JSON.parse(stdout.text)).toEqual([ + { + fileID: 200, + title: "diagram.png", + collection: "Work", + reason: "empty thumbnail (0 bytes)", + }, + ]); + expect(stderr.text).toBe(""); + }); +}); + +describe("backup-metadata --exif", () => { + // Runs the command and returns what it printed to stderr. + const backupMetadata = async (opts: { exif?: boolean; all?: boolean }) => { + expect( + await backupMetadataCommand(context(), join(root, "dump"), opts), + ).toBe(0); + return stderr.text; + }; + + it("--exif extracts EXIF", async () => { + expect(await backupMetadata({ exif: true })).toContain( + "[beach.jpg] Extracting EXIF...\n", + ); + }); + + it("--all extracts EXIF", async () => { + expect(await backupMetadata({ all: true })).toContain( + "[beach.jpg] Extracting EXIF...\n", + ); + }); + + it("without either flag extracts no EXIF", async () => { + expect(await backupMetadata({})).not.toContain("Extracting EXIF"); + }); +}); + +// Each test first runs `collections` so the cache holds the account as it was, +// then changes the server under it. +describe("backup-metadata and the thumbnail helpers refresh first", () => { + beforeEach(async () => { + expect(await collectionsCommand(context(), {})).toBe(0); + stdout.text = ""; + stderr.text = ""; + }); + + it("backup-metadata writes a file added since the cache was written", async () => { + const ctx = context(fakeClient({ withNewFile: true })); + const dir = join(root, "dump"); + expect(await backupMetadataCommand(ctx, dir, {})).toBe(0); + expect( + existsSync(join(dir, "collections", "1-Vacation", "102.json")), + ).toBe(true); + }); + + it("list-missing-thumbnails checks a file added since the cache was written", async () => { + const ctx = context( + fakeClient({ withNewFile: true, emptyThumbID: 102 }), + ); + expect(await listMissingThumbnailsCommand(ctx, {})).toBe(0); + expect(stdout.text).toBe( + "102\tnew.jpg\tVacation\tempty thumbnail (0 bytes)\n", + ); + }); + + it("fix-missing-thumbnails finds a file added since the cache was written", async () => { + const ctx = context(fakeClient({ withNewFile: true })); + expect( + await fixMissingThumbnailsCommand(ctx, { + file: ["102"], + json: true, + }), + ).toBe(0); + // Found, then skipped because the server records no thumbnail size + // for it; a file missing from the cache would fail as not found. + expect(JSON.parse(stdout.text)).toMatchObject([ + { fileID: 102, title: "new.jpg", status: "skipped" }, + ]); + }); + + // `run` in `cli-run.ts` prints a thrown error as one line and exits 1. + it("all three throw when the refresh fails", async () => { + const ctx = context( + fakeClient({ refreshError: "HTTP 503 from server" }), + ); + const dir = join(root, "dump"); + await expect(backupMetadataCommand(ctx, dir, {})).rejects.toThrow( + "HTTP 503 from server", + ); + expect(existsSync(dir)).toBe(false); + await expect(listMissingThumbnailsCommand(ctx, {})).rejects.toThrow( + "HTTP 503 from server", + ); + await expect( + fixMissingThumbnailsCommand(ctx, { file: ["100"] }), + ).rejects.toThrow("HTTP 503 from server"); + expect(stdout.text).toBe(""); + }); +}); diff --git a/test/cli/metadata-backup.test.ts b/test/cli/metadata-backup.test.ts index c13c612..6d597e5 100644 --- a/test/cli/metadata-backup.test.ts +++ b/test/cli/metadata-backup.test.ts @@ -38,7 +38,7 @@ import { join } from "node:path"; import { tmpdir } from "node:os"; import sodium from "libsodium-wrappers-sumo"; import { SRP, SrpServer } from "fast-srp-hap"; -import { beforeAll, afterAll, describe, expect, it } from "vitest"; +import { beforeAll, afterAll, describe, expect, it, vi } from "vitest"; import { init, toBase64, @@ -53,8 +53,16 @@ import { runMetadataBackup, type MetadataBackupOptions, } from "../../src/metadata-backup.js"; +import { backupMetadataCommand } from "../../src/cli-commands.js"; import type { KeyAttributes } from "../../src/auth/types.js"; +// One file per ML data request, so the two files of the mock account are +// fetched in two requests and one of them can fail on its own. +vi.mock("../../src/mldata-fetch.js", async (importOriginal) => ({ + ...(await importOriginal<typeof import("../../src/mldata-fetch.js")>()), + MLDATA_BATCH_SIZE: 1, +})); + const TEST_EMAIL = "metabackup@example.com"; const TEST_PASSWORD = "metapass"; const TEST_OPS = 2; @@ -165,14 +173,15 @@ const buildMetaMock = async (): Promise<MetaMockState> => { }, }; - // Collection 2: "Work" with no magic metadata + // Collection 2: "../Work" with no magic metadata. The server chose a name + // that tries to climb out of the backup directory. const ck2 = sodium.crypto_secretbox_keygen(); const { ciphertext: encCK2, nonce: ck2N } = encryptSecretbox( ck2, masterKey, ); const { ciphertext: encCN2, nonce: cn2N } = encryptSecretbox( - new TextEncoder().encode("Work"), + new TextEncoder().encode("../Work"), ck2, ); const rawColl2 = { @@ -346,7 +355,8 @@ const buildMetaMock = async (): Promise<MetaMockState> => { }; }; -const buildMetaFetch = (m: MetaMockState) => { +// `failMLDataFor`: answer 500 to every ML data request that asks for this file. +const buildMetaFetch = (m: MetaMockState, failMLDataFor?: number) => { let srpServer: SrpServer; return (async ( input: RequestInfo | URL, @@ -402,6 +412,8 @@ const buildMetaFetch = (m: MetaMockState) => { } if (path === "/files/data/fetch") { const body = JSON.parse(init?.body as string); + if ((body.fileIDs as number[]).includes(failMLDataFor!)) + return new Response("server error", { status: 500 }); const data = (body.fileIDs as number[]) .filter((id: number) => m.encryptedMLData[id]) .map((id: number) => ({ @@ -496,7 +508,8 @@ describe("quak backup-metadata", () => { await runBackup(outDir); const collDirs = readdirSync(join(outDir, "collections")); - expect(collDirs.length).toBe(2); + // "../Work" is sanitized into one directory name. + expect(collDirs.sort()).toEqual(["10-Vacation", "20-__Work"]); // Find the Vacation collection dir (prefixed with ID) const vacDir = collDirs.find((d) => d.includes("Vacation"))!; @@ -619,5 +632,78 @@ describe("quak backup-metadata", () => { expect(fileMeta.imageMetadata.format).toBe("jpeg"); expect(fileMeta.imageMetadata.width).toBe(100); expect(fileMeta.imageMetadata.height).toBe(80); + expect(fileMeta.imageMetadataError).toBeUndefined(); + + // File 200 has no original on the mock server, so extraction fails + // and the reason is recorded instead of the field being left out. + const workDir = collDirs.find((d) => d.includes("Work"))!; + const failedMeta = JSON.parse( + readFileSync( + join(outDir, "collections", workDir, "200.json"), + "utf-8", + ), + ); + expect(failedMeta.imageMetadata).toBeUndefined(); + expect(failedMeta.imageMetadataError).toEqual(expect.any(String)); + }); +}); + +describe("quak backup-metadata when an ML data request fails", () => { + // Run the CLI command against the mock and return its exit code, stderr + // and output directory. + const runCommand = async (failMLDataFor?: number) => { + const client = await Client.login({ + email: TEST_EMAIL, + password: TEST_PASSWORD, + apiOptions: { + fetch: buildMetaFetch(mock, failMLDataFor), + retry: { sleep: async () => {} }, + }, + }); + const outDir = mkdtempSync(join(testDir, "ml-fail-")); + let stderr = ""; + const code = await backupMetadataCommand( + { + stdout: { write: () => true }, + stderr: { write: (text: string) => (stderr += text) }, + sessionDir: testDir, + cacheDir: mkdtempSync(join(testDir, "cache-")), + loadSession: () => client, + }, + outDir, + {}, + ); + return { code, stderr, outDir }; + }; + + it("writes every file, marks the failed batch's files, and exits 1", async () => { + const { code, stderr, outDir } = await runCommand(200); + + expect(code).toBe(1); + expect(stderr).toContain("ML data request for 1 file(s) failed"); + + const ok = JSON.parse( + readFileSync( + join(outDir, "collections", "10-Vacation", "100.json"), + "utf-8", + ), + ); + expect(ok.mlData.clip.embedding).toEqual([0.5, 0.6, 0.7]); + expect(ok.mlDataError).toBeUndefined(); + + const failed = JSON.parse( + readFileSync( + join(outDir, "collections", "20-__Work", "200.json"), + "utf-8", + ), + ); + expect(failed.metadata.title).toBe("diagram.png"); + expect(failed.mlData).toBeUndefined(); + expect(failed.mlDataError).toContain("500"); + }); + + it("exits 0 when every ML data request succeeds", async () => { + const { code } = await runCommand(); + expect(code).toBe(0); }); }); diff --git a/test/cli/metadata-exif.test.ts b/test/cli/metadata-exif.test.ts new file mode 100644 index 0000000..4a8cf55 --- /dev/null +++ b/test/cli/metadata-exif.test.ts @@ -0,0 +1,136 @@ +/** + * Tests for the JPEG EXIF scan behind `quak backup-metadata --exif`. + * + * The originals come from users' libraries, so a truncated or corrupt JPEG + * must neither hang the scan nor throw out of it, and a malformed file must be + * told apart from one that simply has no EXIF: the record carries the reason in + * `exifError`. Each input below is a short hand-built byte array. + */ + +import { describe, expect, it } from "vitest"; +import { + extractExifFromJpeg, + extractImageMetadata, +} from "../../src/metadata-backup.js"; + +const SOI = [0xff, 0xd8]; // start of image +const SOS = [0xff, 0xda, 0x00, 0x02]; // start of scan, where the scan stops +const EXIF_HEADER = [0x45, 0x78, 0x69, 0x66, 0x00, 0x00]; // "Exif\0\0" + +// A big-endian TIFF block with one IFD entry: Orientation (0x0112), SHORT, 6. +const TIFF_ORIENTATION_6 = [ + 0x4d, 0x4d, 0x00, 0x2a, 0x00, 0x00, 0x00, 0x08, 0x00, 0x01, 0x01, 0x12, + 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x00, 0x06, 0x00, 0x00, 0x00, 0x00, + 0x00, 0x00, +]; + +// An APP1 segment whose length field matches its data. +const app1 = (data: number[]): number[] => { + const len = data.length + 2; + return [0xff, 0xe1, len >> 8, len & 0xff, ...data]; +}; + +const bytes = (...parts: number[][]): Uint8Array => + new Uint8Array(parts.flat()); + +describe("extractExifFromJpeg", () => { + it("returns the EXIF segment of a valid JPEG", () => { + const data = [...EXIF_HEADER, ...TIFF_ORIENTATION_6]; + const scan = extractExifFromJpeg(bytes(SOI, app1(data), SOS)); + expect(scan.error).toBeUndefined(); + expect([...scan.exif!]).toEqual(data); + }); + + it("returns nothing for a file that is not a JPEG", () => { + const png = bytes([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a]); + expect(extractExifFromJpeg(png)).toEqual({}); + }); + + it("returns nothing for a JPEG without EXIF", () => { + const app0 = [0xff, 0xe0, 0x00, 0x04, 0x00, 0x00]; + expect(extractExifFromJpeg(bytes(SOI, app0, SOS))).toEqual({}); + }); + + it("ignores an APP1 segment too short to hold the Exif header", () => { + // A length under 8 cannot hold the six-byte "Exif\0\0" header, so the + // segment is not EXIF. This one has length 7 and holds only "Exif\0", + // which the old code, lacking the length check, returned as EXIF. + const short = app1(EXIF_HEADER.slice(0, 5)); + expect(extractExifFromJpeg(bytes(SOI, short, SOS))).toEqual({}); + }); + + it("accepts an APP1 segment of length 8 holding just the Exif header", () => { + const scan = extractExifFromJpeg(bytes(SOI, app1(EXIF_HEADER), SOS)); + expect(scan.error).toBeUndefined(); + expect([...scan.exif!]).toEqual(EXIF_HEADER); + }); + + it("reports a JPEG truncated inside a segment header", () => { + const scan = extractExifFromJpeg(bytes(SOI, [0xff, 0xe1, 0x00])); + expect(scan.exif).toBeUndefined(); + expect(scan.error).toMatch(/truncated segment length/); + }); + + it("reports a JPEG that ends before the image data", () => { + const app0 = [0xff, 0xe0, 0x00, 0x04, 0x00, 0x00]; + const scan = extractExifFromJpeg(bytes(SOI, app0)); + expect(scan.error).toMatch(/ends before the image data/); + }); + + it("stops on a zero-length segment instead of looping", () => { + // A length of 0 would otherwise step the scan by 2 bytes at a time + // through the rest of the file, reading garbage as markers. + const zero = [0xff, 0xe0, 0x00, 0x00]; + const scan = extractExifFromJpeg( + bytes(SOI, zero, zero, zero, zero, SOS), + ); + expect(scan.error).toMatch(/segment length 0 at byte 2 is too small/); + }); + + it("stops on a segment length of 1", () => { + const scan = extractExifFromJpeg( + bytes(SOI, [0xff, 0xe0, 0x00, 0x01], SOS), + ); + expect(scan.error).toMatch(/segment length 1 at byte 2 is too small/); + }); + + it("reports a segment length that runs past the end of the file", () => { + // APP1 claims 0x4000 bytes but only the "Exif\0\0" header follows. + const scan = extractExifFromJpeg( + bytes(SOI, [0xff, 0xe1, 0x40, 0x00], EXIF_HEADER), + ); + expect(scan.exif).toBeUndefined(); + expect(scan.error).toMatch(/runs past the end of the file/); + }); +}); + +describe("extractImageMetadata", () => { + it("parses EXIF from a valid JPEG", () => { + const meta = extractImageMetadata( + bytes(SOI, app1([...EXIF_HEADER, ...TIFF_ORIENTATION_6]), SOS), + ); + expect(meta?.exifError).toBeUndefined(); + expect(meta?.exif).toMatchObject({ Image: { Orientation: 6 } }); + }); + + it("returns nothing for a file that is not a JPEG", () => { + const text = new TextEncoder().encode("just some text, not an image"); + expect(extractImageMetadata(text)).toBeUndefined(); + }); + + it("records the reason when the JPEG is malformed", () => { + const meta = extractImageMetadata( + bytes(SOI, [0xff, 0xe1, 0x40, 0x00], EXIF_HEADER), + ); + expect(meta?.exif).toBeUndefined(); + expect(meta?.exifError).toMatch(/runs past the end of the file/); + }); + + it("keeps the raw bytes and the reason when EXIF cannot be parsed", () => { + const data = [...EXIF_HEADER, 0x58, 0x58]; + const meta = extractImageMetadata(bytes(SOI, app1(data), SOS)); + expect(meta?.exif).toBeUndefined(); + expect(meta?.exifRaw).toBe(Buffer.from(data).toString("base64")); + expect(meta?.exifError).toEqual(expect.any(String)); + }); +}); diff --git a/test/cli/output.test.ts b/test/cli/output.test.ts index 3f5a718..ce03e30 100644 --- a/test/cli/output.test.ts +++ b/test/cli/output.test.ts @@ -64,6 +64,24 @@ describe("CLI file output (issue #52)", () => { expect(thumbnailName(renamedFile)).toBe(`thumb_${RAW_TITLE}`); }); + it("sanitizes the title when naming `quak get` downloads", () => { + // Without `--out`, the server-supplied title names the file, so it must + // not be able to point outside the working directory. + const hostile = { + ...renamedFile, + metadata: { ...renamedFile.metadata, title: "../../.bashrc" }, + }; + expect(originalName(hostile)).toBe("__.._.bashrc"); + expect(thumbnailName(hostile)).toBe("thumb___.._.bashrc"); + + const untitled = { + ...renamedFile, + metadata: { ...renamedFile.metadata, title: "" }, + }; + expect(originalName(untitled)).toBe("file-100"); + expect(thumbnailName(untitled)).toBe("thumb_file-100"); + }); + it("does not use the editedName/editedTime projection", () => { const record = deriveRecords([], [renamedFile]).photos.get(100); // The projection prefers the edits and reports milliseconds; the CLI diff --git a/test/cli/run.test.ts b/test/cli/run.test.ts new file mode 100644 index 0000000..29d0a92 --- /dev/null +++ b/test/cli/run.test.ts @@ -0,0 +1,58 @@ +/** + * Tests for `run` in `src/cli-run.ts`, which every CLI command goes through. + */ + +import { PassThrough } from "node:stream"; +import { describe, it, expect } from "vitest"; + +import { run } from "../../src/cli-run.js"; + +// A stream whose written text is kept in `text`; writes finish at once, so +// nothing is left waiting to drain. +const collector = (): { stream: PassThrough; text: () => string } => { + const stream = new PassThrough(); + const chunks: string[] = []; + stream.on("data", (chunk: Buffer) => chunks.push(chunk.toString())); + return { stream, text: () => chunks.join("") }; +}; + +const runToExit = async ( + command: Promise<number>, +): Promise<{ code: number; stdout: string; stderr: string }> => { + const stdout = collector(); + const stderr = collector(); + const code = await new Promise<number>((resolve) => { + void run(command, stdout.stream, stderr.stream, resolve); + }); + return { code, stdout: stdout.text(), stderr: stderr.text() }; +}; + +describe("run", () => { + it("exits with the code the command returns", async () => { + const result = await runToExit(Promise.resolve(3)); + expect(result).toEqual({ code: 3, stdout: "", stderr: "" }); + }); + + it("prints a thrown error as one line without a stack trace and exits 1", async () => { + const failing = async (): Promise<number> => { + throw new Error( + "ENOTDIR: not a directory, mkdir '/dev/null/x/originals'", + ); + }; + const result = await runToExit(failing()); + expect(result.code).toBe(1); + expect(result.stdout).toBe(""); + expect(result.stderr).toBe( + "quak: ENOTDIR: not a directory, mkdir '/dev/null/x/originals'\n", + ); + }); + + it("prints a thrown value that is not an Error", async () => { + const result = await runToExit(Promise.reject("offline")); + expect(result).toEqual({ + code: 1, + stdout: "", + stderr: "quak: offline\n", + }); + }); +}); diff --git a/test/client/session.test.ts b/test/client/session.test.ts new file mode 100644 index 0000000..7aa2359 --- /dev/null +++ b/test/client/session.test.ts @@ -0,0 +1,197 @@ +/** + * Tests for the client session lifecycle: `toJSON`, `fromJSON`, `logout`, and + * the CLI's `loadSession`, which reads the saved session file back into a + * client. + */ + +import { mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import sodium from "libsodium-wrappers-sumo"; +import { afterAll, beforeAll, describe, expect, it } from "vitest"; +import { init, toBase64 } from "../../src/crypto/index.js"; +import { Client, type ClientSnapshot } from "../../src/client.js"; +import { loadSession } from "../../src/cli-session.js"; + +const validSnapshot = (): ClientSnapshot => { + const kp = sodium.crypto_box_keypair(); + return { + email: "user@example.com", + userID: 42, + token: "test-token", + masterKey: toBase64(sodium.crypto_secretbox_keygen()), + secretKey: toBase64(kp.privateKey), + publicKey: toBase64(kp.publicKey), + }; +}; + +// The client's key buffers are private; the tests read them to prove that +// logout wipes them. +const keyBuffers = (client: Client): Uint8Array[] => [ + client["masterKey"], + client["secretKey"], + client["publicKey"], +]; + +beforeAll(async () => { + await init(); +}); + +describe("Client.toJSON", () => { + it("round-trips through fromJSON unchanged", () => { + const snapshot = validSnapshot(); + expect(Client.fromJSON(snapshot).toJSON()).toEqual(snapshot); + }); + + it("throws instead of emitting a snapshot without a token", () => { + const client = Client.fromJSON(validSnapshot()); + client.getApiClient().clearAuthToken(); + expect(() => client.toJSON()).toThrow(/no auth token/); + }); +}); + +describe("Client.fromJSON", () => { + const shortKey = toBase64(new Uint8Array(16)); + + it.each([ + ["email", undefined], + ["email", 7], + ["email", ""], + ["token", undefined], + ["token", null], + ["token", ""], + ["userID", undefined], + ["userID", "42"], + ["userID", 4.2], + ["masterKey", undefined], + ["masterKey", 7], + ["masterKey", "not base64!"], + ["masterKey", shortKey], + ["secretKey", undefined], + ["secretKey", "not base64!"], + ["secretKey", shortKey], + ["publicKey", undefined], + ["publicKey", "not base64!"], + ["publicKey", shortKey], + ])("rejects %s = %j, naming the field", (field, value) => { + const snapshot: Record<string, unknown> = { ...validSnapshot() }; + snapshot[field] = value; + expect(() => Client.fromJSON(snapshot)).toThrow( + new RegExp(`^Invalid session data: ${field} `), + ); + }); + + it.each([null, "a string", 42])("rejects a non-object %j", (value) => { + expect(() => Client.fromJSON(value)).toThrow( + /^Invalid session data: not a JSON object/, + ); + }); +}); + +describe("Client.logout", () => { + it("zeroes the key buffers and clears the token", () => { + const client = Client.fromJSON(validSnapshot()); + const api = client.getApiClient(); + const keys = keyBuffers(client); + + client.logout(); + + for (const key of keys) { + expect(key.length).toBe(32); + expect(key.every((b) => b === 0)).toBe(true); + } + expect(api.getAuthToken()).toBeUndefined(); + }); + + it("makes every later operation throw", async () => { + const client = Client.fromJSON(validSnapshot()); + client.logout(); + + expect(() => client.whoami()).toThrow(/logged out/); + expect(() => client.toJSON()).toThrow(/logged out/); + expect(() => client.getApiClient()).toThrow(/logged out/); + expect(() => client.contentSource()).toThrow(/logged out/); + await expect(client.listCollections()).rejects.toThrow(/logged out/); + await expect(client.collectionsSince({ sinceTime: 0 })).rejects.toThrow( + /logged out/, + ); + await expect( + client.filesSince({ + collectionID: 1, + collectionKey: new Uint8Array(32), + sinceTime: 0, + }), + ).rejects.toThrow(/logged out/); + await expect( + client.fetchMLData({ fileIDs: [1], fileKeys: new Map() }), + ).rejects.toThrow(/logged out/); + }); + + it("stops a listing in flight from decrypting with the zeroed keys", async () => { + // The server answers only after the client has logged out. If the + // listing went on to decrypt this row with all-zero keys it would fail + // with a decryption error, not the logged-out one. + const row = { + id: 1, + owner: { id: 42 }, + encryptedKey: toBase64(new Uint8Array(48)), + keyDecryptionNonce: toBase64(new Uint8Array(24)), + updationTime: 1, + }; + const client: Client = Client.fromJSON(validSnapshot(), { + fetch: async () => { + client.logout(); + return new Response(JSON.stringify({ collections: [row] }), { + status: 200, + headers: { "content-type": "application/json" }, + }); + }, + }); + + await expect(client.listCollections()).rejects.toThrow(/logged out/); + }); +}); + +describe("loadSession", () => { + let dir: string; + + beforeAll(() => { + dir = mkdtempSync(join(tmpdir(), "quak-session-test-")); + }); + + afterAll(() => { + rmSync(dir, { recursive: true, force: true }); + }); + + it("returns null when there is no session file", () => { + expect(loadSession(join(dir, "missing.json"))).toBeNull(); + }); + + it("restores a client from a valid session file", () => { + const path = join(dir, "valid.json"); + writeFileSync(path, JSON.stringify(validSnapshot())); + expect(loadSession(path)!.whoami()).toEqual({ + email: "user@example.com", + userID: 42, + }); + }); + + it("says the file is corrupt when it is not JSON", () => { + const path = join(dir, "truncated.json"); + writeFileSync(path, '{"email": "user@exa'); + expect(() => loadSession(path)).toThrow( + `Session file ${path} is corrupt`, + ); + }); + + it("says the file is corrupt and names the bad field", () => { + const path = join(dir, "bad-key.json"); + writeFileSync( + path, + JSON.stringify({ ...validSnapshot(), secretKey: "AAAA" }), + ); + expect(() => loadSession(path)).toThrow( + new RegExp(`^Session file ${path} is corrupt: .*secretKey`), + ); + }); +}); diff --git a/test/crypto/hash.test.ts b/test/crypto/hash.test.ts new file mode 100644 index 0000000..153ec0a --- /dev/null +++ b/test/crypto/hash.test.ts @@ -0,0 +1,33 @@ +import { beforeAll, describe, expect, it } from "vitest"; +import { + chunkHashFinal, + chunkHashInit, + chunkHashUpdate, + init, +} from "../../src/crypto/index.js"; + +beforeAll(async () => { + await init(); +}); + +describe("content hash", () => { + // RFC 7693 Appendix A: BLAKE2b-512 of "abc". + const abc = Buffer.from( + "ba80a53f981c4d0d6a2797b69f12f6e94c212f14685ac4b74b12bb6fdbffa2d1" + + "7d87c5392aab792dc252d5de4533cc9518d38aa8dbf1925ab92386edd4009923", + "hex", + ).toString("base64"); + + it("is unkeyed BLAKE2b-512 in standard base64", () => { + const state = chunkHashInit(); + chunkHashUpdate(state, new TextEncoder().encode("abc")); + expect(chunkHashFinal(state)).toBe(abc); + }); + + it("gives the same hash when the input arrives in chunks", () => { + const state = chunkHashInit(); + chunkHashUpdate(state, new TextEncoder().encode("a")); + chunkHashUpdate(state, new TextEncoder().encode("bc")); + expect(chunkHashFinal(state)).toBe(abc); + }); +}); diff --git a/test/crypto/kdf.test.ts b/test/crypto/kdf.test.ts index 670d80e..d252653 100644 --- a/test/crypto/kdf.test.ts +++ b/test/crypto/kdf.test.ts @@ -29,10 +29,10 @@ describe("crypto.deriveKEK (Argon2id)", () => { }); /** - * Cheap parameters used so the test suite stays under the 30-second - * budget. The real production parameters Ente uses are larger - * (memLimit up to 1 GiB, opsLimit 3-16). The algorithm is the same - * regardless of parameters. + * Cheap parameters used so the test suite stays under the 90-second + * `timeout` in the `test` phase of the `Dockerfile`. The real production + * parameters Ente uses are larger (memLimit up to 1 GiB, opsLimit 3-16). + * The algorithm is the same regardless of parameters. */ const TEST_OPS = 2; const TEST_MEM = 64 * 1024 * 1024; // 64 MiB diff --git a/test/download/download.test.ts b/test/download/download.test.ts index eb27c65..f9e144d 100644 --- a/test/download/download.test.ts +++ b/test/download/download.test.ts @@ -48,7 +48,9 @@ */ import { + chmodSync, existsSync, + mkdirSync, readdirSync, readFileSync, rmSync, @@ -59,6 +61,7 @@ import { dirname, join } from "node:path"; import { tmpdir } from "node:os"; import { createHash } from "node:crypto"; import sodium from "libsodium-wrappers-sumo"; +import { zipSync } from "fflate"; import { beforeAll, beforeEach, @@ -186,7 +189,31 @@ vi.mock("node:fs/promises", async (importOriginal) => { }; }); +/** + * `chunkHashUpdate` is wrapped to record the length of every piece hashed, so + * a test can show that a live photo entry reaches the hash in pieces far + * smaller than the entry, rather than decompressed whole first. + */ +const hashHook = vi.hoisted(() => ({ + lengths: [] as number[], +})); + +vi.mock("../../src/crypto/index.js", async (importOriginal) => { + const actual = + await importOriginal<typeof import("../../src/crypto/index.js")>(); + return { + ...actual, + chunkHashUpdate: ( + ...args: Parameters<typeof actual.chunkHashUpdate> + ): void => { + hashHook.lengths.push(args[1].length); + actual.chunkHashUpdate(...args); + }, + }; +}); + beforeEach(() => { + hashHook.lengths.length = 0; renameHook.calls.length = 0; renameHook.failWith = null; durabilityHook.events.length = 0; @@ -221,7 +248,8 @@ afterAll(() => { * `sodium.randombytes_buf` goes through the wasm wrapper a byte at a time and * costs roughly 20 seconds for the 4 MiB chunk below — about two hundred * times what it costs to encrypt the same buffer, and on its own enough to - * push `make test` past the 30-second cap in `script/test`. This loop fills + * push `make test` past the 90-second `timeout` in the `test` phase of the + * `Dockerfile`. This loop fills * 4 MiB in a few milliseconds. */ const patternBytes = (length: number, seed: number): Uint8Array => { @@ -501,6 +529,22 @@ const entryPoints = [ { name: "downloadThumbnail", download: downloadThumbnail }, ]; +// With no `outPath`, the destination is named after `metadata.title`, relative +// to the working directory. Such tests run inside a temporary directory: +// `make check` must not create files in the repo root. +const inDirectory = async <T>( + dir: string, + run: () => Promise<T>, +): Promise<T> => { + const previous = process.cwd(); + process.chdir(dir); + try { + return await run(); + } finally { + process.chdir(previous); + } +}; + // --------------------------------------------------------------------------- // Tests // --------------------------------------------------------------------------- @@ -536,26 +580,59 @@ describe("downloadFile", () => { }); it("uses metadata.title as filename when outPath is omitted", async () => { - // With no `outPath`, the destination is `metadata.title`, used - // verbatim as a path. The title here is therefore given inside the - // test's temporary directory: a bare relative name would resolve - // against the process working directory, i.e. the repo root, and - // `make check` must not create files in the repo — a failure between - // the write and any cleanup would leave one behind. const plaintext = new Uint8Array([1, 2, 3]); const key = sodium.crypto_secretstream_xchacha20poly1305_keygen(); const { header, ciphertext } = encryptFileBody(plaintext, key); const thumbPush = sodium.crypto_secretstream_xchacha20poly1305_init_push(key); const file = buildMockEnteFile(key, header, thumbPush.header); - const titlePath = join(testDir, "fallback-name.png"); - file.metadata.title = titlePath; + file.metadata.title = "fallback-name.png"; + const dir = mkdtempSync(join(testDir, "title-")); const api = new ApiClient({ fetch: mockFetchForBody(ciphertext) }); - const result = await downloadFile(api, file); + const result = await inDirectory(dir, () => downloadFile(api, file)); - expect(result.path).toBe(titlePath); - expect(readFileSync(result.path)).toEqual(Buffer.from(plaintext)); + expect(result.path).toBe("fallback-name.png"); + expect(readFileSync(join(dir, "fallback-name.png"))).toEqual( + Buffer.from(plaintext), + ); + }); + + it("keeps a hostile title inside the working directory", async () => { + // The server controls the title. `../escaped.png` must not write to + // the parent directory; it becomes one file name in the current one. + const { api, file } = fixtureFor( + multiChunkKey, + multiChunk.header, + multiChunk.body, + ); + file.metadata.title = "../escaped.png"; + const parent = mkdtempSync(join(testDir, "hostile-")); + const dir = join(parent, "cwd"); + mkdirSync(dir); + + const result = await inDirectory(dir, () => downloadFile(api, file)); + + expect(result.path).toBe("__escaped.png"); + expect(readdirSync(dir)).toEqual(["__escaped.png"]); + expect(readdirSync(parent)).toEqual(["cwd"]); + }); + + it("uses an explicit outPath verbatim, even one with ..", async () => { + // The caller is trusted: its path is not sanitized. + const { api, file } = fixtureFor( + multiChunkKey, + multiChunk.header, + multiChunk.body, + ); + const dir = mkdtempSync(join(testDir, "explicit-")); + mkdirSync(join(dir, "sub")); + const outPath = join(dir, "sub", "..", "explicit.bin"); + + const result = await downloadFile(api, file, outPath); + + expect(result.path).toBe(outPath); + expect(existsSync(join(dir, "explicit.bin"))).toBe(true); }); it("handles a larger single-chunk file (random binary payload)", async () => { @@ -617,6 +694,23 @@ describe("downloadThumbnail", () => { expect(result).toEqual({ path: outPath, bytesWritten: 4 }); expect(readFileSync(outPath)).toEqual(Buffer.from(plaintext)); }); + + it("names the thumbnail thumb_ plus the sanitized title", async () => { + const { api, file } = fixtureFor( + multiChunkKey, + multiChunk.header, + multiChunk.body, + ); + file.metadata.title = "/etc/passwd"; + const dir = mkdtempSync(join(testDir, "thumb-title-")); + + const result = await inDirectory(dir, () => + downloadThumbnail(api, file), + ); + + expect(result.path).toBe("thumb__etc_passwd"); + expect(readdirSync(dir)).toEqual(["thumb__etc_passwd"]); + }); }); // --------------------------------------------------------------------------- @@ -922,6 +1016,47 @@ describe.each(entryPoints)( expect(readFileSync(outPath)).toEqual(Buffer.from(existing)); expect(readdirSync(dir)).toEqual(["rename-fails.bin"]); }); + + it("fails without creating anything when the destination directory does not exist", async () => { + const key = sodium.crypto_secretstream_xchacha20poly1305_keygen(); + const { header, ciphertext } = encryptFileBody( + patternBytes(64, 33), + key, + ); + const { api, file } = fixtureFor(key, header, ciphertext); + const dir = freshDir(); + const outPath = join(dir, "missing", "never.bin"); + + await expect(download(api, file, outPath)).rejects.toMatchObject({ + code: "ENOENT", + }); + + // The missing directory is not created on the caller's behalf. + expect(readdirSync(dir)).toEqual([]); + }); + + // Root ignores directory permissions, so this fails when run as root. + // The `test` phase of the `Dockerfile` runs as the `node` user. + it("fails without creating anything when the destination directory is not writable", async () => { + const key = sodium.crypto_secretstream_xchacha20poly1305_keygen(); + const { header, ciphertext } = encryptFileBody( + patternBytes(64, 34), + key, + ); + const { api, file } = fixtureFor(key, header, ciphertext); + const dir = freshDir(); + const outPath = join(dir, "never.bin"); + chmodSync(dir, 0o500); + try { + await expect( + download(api, file, outPath), + ).rejects.toMatchObject({ code: "EACCES" }); + } finally { + chmodSync(dir, 0o700); + } + + expect(readdirSync(dir)).toEqual([]); + }); }, ); @@ -1224,6 +1359,102 @@ describe("download retries: corruption is not retried", () => { expect(requests()).toBe(1); }); + + it("cancels the response body when decryption fails", async () => { + // A backup run carries on past a failed file, so a body left open on + // failure would hold its connection until garbage collection, once + // per failed file. This body delivers a corrupt chunk and then stays + // open, so only a cancel from the downloader can close it. + const corrupted = Uint8Array.from(multiChunk.body); + corrupted[10] ^= 0xff; + let cancelled = false; + const fetch = (async () => + new Response( + new ReadableStream<Uint8Array>({ + start(controller) { + controller.enqueue(corrupted); + }, + cancel() { + cancelled = true; + }, + }), + { status: 200 }, + )) as typeof globalThis.fetch; + const api = new ApiClient({ fetch, retry: { ...noWait, attempts: 1 } }); + const file = buildMockEnteFile( + multiChunkKey, + multiChunk.header, + multiChunk.header, + ); + const outPath = join(mkdtempSync(join(testDir, "cancel-")), "c.bin"); + + await expect(downloadFile(api, file, outPath)).rejects.toThrow( + /authentication failed/i, + ); + + expect(cancelled).toBe(true); + }); + + it("cancels the response body when the temp file cannot be opened", async () => { + // The download fails before a byte of the body is read, so the body + // is still open and only a cancel from the downloader can close it. + let cancelled = false; + const fetch = (async () => + new Response( + new ReadableStream<Uint8Array>({ + start(controller) { + controller.enqueue(multiChunk.body); + }, + cancel() { + cancelled = true; + }, + }), + { status: 200 }, + )) as typeof globalThis.fetch; + const api = new ApiClient({ fetch, retry: { ...noWait, attempts: 1 } }); + const file = buildMockEnteFile( + multiChunkKey, + multiChunk.header, + multiChunk.header, + ); + const outPath = join( + mkdtempSync(join(testDir, "cancel-")), + "missing", + "c.bin", + ); + + await expect(downloadFile(api, file, outPath)).rejects.toMatchObject({ + code: "ENOENT", + }); + + expect(cancelled).toBe(true); + }); + + it("cancels the response body when the header is malformed", async () => { + // The header is rejected before the body is read, so the body is + // still open and only a cancel from the downloader can close it. + let cancelled = false; + const fetch = (async () => + new Response( + new ReadableStream<Uint8Array>({ + start(controller) { + controller.enqueue(multiChunk.body); + }, + cancel() { + cancelled = true; + }, + }), + { status: 200 }, + )) as typeof globalThis.fetch; + const api = new ApiClient({ fetch, retry: { ...noWait, attempts: 1 } }); + const shortHeader = multiChunk.header.subarray(0, 5); + const file = buildMockEnteFile(multiChunkKey, shortHeader, shortHeader); + const outPath = join(mkdtempSync(join(testDir, "cancel-")), "c.bin"); + + await expect(downloadFile(api, file, outPath)).rejects.toThrow(); + + expect(cancelled).toBe(true); + }); }); // --------------------------------------------------------------------------- @@ -1329,7 +1560,13 @@ describe("writeAtomic", () => { // directory so that new entry is on disk too. Do the directory fsync // before the rename, or skip it, and a crash can lose the rename. expect(durabilityHook.events).toHaveLength(3); - expect(durabilityHook.events[0]).toMatch(/^sync:w:.*\.tmp$/); + // The temp name carries this process's ID, so a library opening the + // same cache can tell a write in progress from a leftover. + const tempSync = durabilityHook.events[0]!; + expect(tempSync.startsWith(`sync:w:${dir}/`)).toBe(true); + expect(tempSync.slice(`sync:w:${dir}/`.length)).toMatch( + new RegExp(`^\\.quak-${process.pid}-[0-9a-f]{32}\\.tmp$`), + ); expect(durabilityHook.events[1]).toBe(`rename:${dest}`); expect(durabilityHook.events[2]).toBe(`sync:r:${dir}`); }); @@ -1404,3 +1641,151 @@ describe.each(entryPoints)("$name progress", ({ name, download }) => { expectSameBytes(readFileSync(outPath), plaintext); }); }); + +describe("downloadFile content hash", () => { + // Node's own BLAKE2b-512 is the reference, so these tests do not depend + // on the code under test to compute what they expect. + const blake2b = (bytes: Uint8Array): string => + createHash("blake2b512").update(bytes).digest("base64"); + + // Serve `plaintext` encrypted as file 999 with the given metadata. Four + // responses are scripted so a retried mismatch would show in `requests`. + const setup = (plaintext: Uint8Array, metadata: Partial<FileMetadata>) => { + const key = sodium.crypto_secretstream_xchacha20poly1305_keygen(); + const { header, ciphertext } = encryptFileBody(plaintext, key); + const file = buildMockEnteFile(key, header, header); + file.metadata = { ...file.metadata, ...metadata }; + const body = { kind: "body", bytes: ciphertext } as const; + const { fetch, requests } = scriptedCdnFetch(body, body, body, body); + const api = new ApiClient({ fetch, retry: { ...noWait, attempts: 4 } }); + const dir = mkdtempSync(join(testDir, "hash-")); + const outPath = join(dir, "f.bin"); + return { + run: () => downloadFile(api, file, outPath), + dir, + outPath, + requests, + }; + }; + + const livePhotoZip = zipSync({ + "image.heic": patternBytes(500, 81), + "video.mov": patternBytes(900, 82), + }); + const livePhotoHash = `${blake2b(patternBytes(500, 81))}:${blake2b(patternBytes(900, 82))}`; + + it("stores a file whose hash matches", async () => { + const plaintext = patternBytes(700, 80); + const t = setup(plaintext, { hash: blake2b(plaintext) }); + + await t.run(); + + expectSameBytes(readFileSync(t.outPath), plaintext); + }); + + it("rejects a mismatch, stores nothing, names the file and does not retry", async () => { + const t = setup(patternBytes(700, 80), { + hash: blake2b(patternBytes(700, 79)), + }); + + await expect(t.run()).rejects.toThrow( + /file 999: content hash .* does not match/, + ); + + expect(readdirSync(t.dir)).toEqual([]); + expect(t.requests()).toBe(1); + }); + + it("stores a file with no recorded hash unchecked", async () => { + const plaintext = patternBytes(700, 80); + const t = setup(plaintext, { hash: undefined }); + + await t.run(); + + expectSameBytes(readFileSync(t.outPath), plaintext); + }); + + it("stores a live photo whose image and video hashes match", async () => { + const t = setup(livePhotoZip, { + fileType: "livePhoto", + hash: livePhotoHash, + }); + + await t.run(); + + expectSameBytes(readFileSync(t.outPath), livePhotoZip); + }); + + it("hashes a large live photo entry as it decompresses, never whole", async () => { + // 64 MiB of zeros deflates to a few kilobytes, the shape of a ZIP + // that would exhaust memory if expanded whole. + const image = new Uint8Array(64 * 1024 * 1024); + const video = patternBytes(900, 83); + const zip = zipSync({ "image.heic": image, "video.mov": video }); + const t = setup(zip, { + fileType: "livePhoto", + hash: `${blake2b(image)}:${blake2b(video)}`, + }); + + await t.run(); + + expectSameBytes(readFileSync(t.outPath), zip); + const hashed = hashHook.lengths.reduce((a, b) => a + b, 0); + expect(hashed).toBe(image.length + video.length); + expect(Math.max(...hashHook.lengths)).toBeLessThanOrEqual( + 2 * STREAM_CHUNK_SIZE, + ); + }); + + it("rejects a live photo whose hash does not match", async () => { + // The whole ZIP's hash is not the recorded one: each part is hashed. + const t = setup(livePhotoZip, { + fileType: "livePhoto", + hash: blake2b(livePhotoZip), + }); + + await expect(t.run()).rejects.toThrow( + /file 999: content hash .* does not match/, + ); + + expect(readdirSync(t.dir)).toEqual([]); + }); + + it("rejects a live photo that is not a readable ZIP and does not retry", async () => { + // Bytes 8-9 of a ZIP entry's local header name its compression + // method; 99 is one no reader knows, so the entry cannot be read. + const zip = livePhotoZip.slice(); + zip[8] = 99; + zip[9] = 0; + const t = setup(zip, { fileType: "livePhoto", hash: livePhotoHash }); + + await expect(t.run()).rejects.toThrow( + /file 999: live photo is not a readable ZIP/, + ); + + expect(readdirSync(t.dir)).toEqual([]); + expect(t.requests()).toBe(1); + }); + + it("rejects a live photo ZIP with no image entry", async () => { + const zip = zipSync({ "video.mov": patternBytes(900, 82) }); + const t = setup(zip, { fileType: "livePhoto", hash: livePhotoHash }); + + await expect(t.run()).rejects.toThrow( + /file 999: live photo ZIP does not hold both an image and a video/, + ); + + expect(readdirSync(t.dir)).toEqual([]); + }); + + it("rejects a live photo ZIP with no video entry", async () => { + const zip = zipSync({ "image.heic": patternBytes(500, 81) }); + const t = setup(zip, { fileType: "livePhoto", hash: livePhotoHash }); + + await expect(t.run()).rejects.toThrow( + /file 999: live photo ZIP does not hold both an image and a video/, + ); + + expect(readdirSync(t.dir)).toEqual([]); + }); +}); diff --git a/test/filename/filename.test.ts b/test/filename/filename.test.ts new file mode 100644 index 0000000..81e7c56 --- /dev/null +++ b/test/filename/filename.test.ts @@ -0,0 +1,93 @@ +// File names built from server-supplied metadata. +// +// quak does not trust the server. A file's title and a collection's name are +// decrypted from data the server hands us, and a hostile server (or a +// compromised account) can set them to anything. quak uses them to name files +// on disk: `quak get` without `--out`, `downloadFile` without `outPath`, the +// backup's symlink and collection directories, and the extension of every file +// in the originals cache. Each of those goes through `sanitizeFileName` or +// `safeExtension`, so a title can only ever name one file inside the directory +// the caller chose. +// +// A path the user supplies (`--out`, `outPath`) is never sanitized: the caller +// is trusted, the server is not. + +import { describe, expect, it } from "vitest"; + +import { safeExtension, sanitizeFileName } from "../../src/filename.js"; + +const FALLBACK = "file-42"; + +describe("sanitizeFileName", () => { + it("passes a normal title through unchanged", () => { + expect(sanitizeFileName("IMG_0001.HEIC", FALLBACK)).toBe( + "IMG_0001.HEIC", + ); + expect(sanitizeFileName("Holiday 2024 (1).jpg", FALLBACK)).toBe( + "Holiday 2024 (1).jpg", + ); + expect(sanitizeFileName("café.jpg", FALLBACK)).toBe("café.jpg"); + }); + + it("cannot climb out of the directory with ../", () => { + // Without sanitizing, this would overwrite the user's SSH keys. + expect(sanitizeFileName("../../.ssh/authorized_keys", FALLBACK)).toBe( + "__.._.ssh_authorized_keys", + ); + expect(sanitizeFileName("..", FALLBACK)).toBe("_"); + expect(sanitizeFileName("..\\..\\x", FALLBACK)).toBe("__.._x"); + }); + + it("cannot name an absolute path", () => { + expect(sanitizeFileName("/etc/passwd", FALLBACK)).toBe("_etc_passwd"); + expect(sanitizeFileName("C:\\Windows\\x.dll", FALLBACK)).toBe( + "C__Windows_x.dll", + ); + }); + + it("replaces embedded separators, so the name stays one file", () => { + expect(sanitizeFileName("a/b\\c.jpg", FALLBACK)).toBe("a_b_c.jpg"); + }); + + it("replaces NUL and other control characters", () => { + // A NUL truncates the path in C code and makes Node's fs throw. + expect(sanitizeFileName("evil\0.jpg", FALLBACK)).toBe("evil_.jpg"); + expect(sanitizeFileName("line\nbreak\x7f.jpg", FALLBACK)).toBe( + "line_break_.jpg", + ); + }); + + it("does not produce a hidden file", () => { + expect(sanitizeFileName(".bashrc", FALLBACK)).toBe("_bashrc"); + }); + + it("does not produce a Windows device name", () => { + expect(sanitizeFileName("CON", FALLBACK)).toBe("_CON"); + expect(sanitizeFileName("nul.txt", FALLBACK)).toBe("_nul.txt"); + expect(sanitizeFileName("LPT1", FALLBACK)).toBe("_LPT1"); + // Only the exact names are reserved. + expect(sanitizeFileName("console.jpg", FALLBACK)).toBe("console.jpg"); + }); + + it("falls back to the given name for an empty title", () => { + expect(sanitizeFileName("", FALLBACK)).toBe(FALLBACK); + }); +}); + +describe("safeExtension", () => { + it("keeps a normal extension", () => { + expect(safeExtension("IMG_0001.HEIC")).toBe(".HEIC"); + expect(safeExtension("clip.mp4")).toBe(".mp4"); + }); + + it("uses .bin when there is no extension", () => { + expect(safeExtension("")).toBe(".bin"); + expect(safeExtension("README")).toBe(".bin"); + }); + + it("uses .bin when the extension holds anything but letters and digits", () => { + expect(safeExtension("x.j\\..\\pg")).toBe(".bin"); + expect(safeExtension("x.jp g")).toBe(".bin"); + expect(safeExtension("x.jpg\0")).toBe(".bin"); + }); +}); diff --git a/test/library/content-library.test.ts b/test/library/content-library.test.ts index a89f47c..fe60236 100644 --- a/test/library/content-library.test.ts +++ b/test/library/content-library.test.ts @@ -116,7 +116,7 @@ describe("Library content wiring", () => { expect(lib.photos.byID({ fileID: 1 })!.record().thumbnailPath).toBe( result.path, ); - lib.close(); + await lib.close(); }); it("drives thumbnails.ensure through the cache", async () => { @@ -139,7 +139,7 @@ describe("Library content wiring", () => { expect(results).toEqual([ { fileID: 1, path: join(root, "cache", "thumbnails", "1.jpg") }, ]); - lib.close(); + await lib.close(); }); it("throws from content methods when opened without a content source", async () => { @@ -155,6 +155,6 @@ describe("Library content wiring", () => { await expect( lib.thumbnails.ensure({ fileIDs: [1], priority: "visible" }), ).rejects.toThrow(/content cache/i); - lib.close(); + await lib.close(); }); }); diff --git a/test/library/content.test.ts b/test/library/content.test.ts index 6fe91ab..5db599c 100644 --- a/test/library/content.test.ts +++ b/test/library/content.test.ts @@ -33,6 +33,7 @@ import { mkdirSync, statSync, } from "node:fs"; +import { spawnSync } from "node:child_process"; import { tmpdir } from "node:os"; import { join } from "node:path"; @@ -161,15 +162,28 @@ describe("ContentCache.open", () => { expect(statSync(thumbnails).mode & 0o777).toBe(0o700); }); - it("reaps orphan temp files but keeps complete content", async () => { + it("removes temp files of an exited process, keeping those of a running one and complete content", async () => { const originals = join(cacheDir, "originals"); const thumbnails = join(cacheDir, "thumbnails"); mkdirSync(originals, { recursive: true }); mkdirSync(thumbnails, { recursive: true }); - const orphan = join(originals, ".quak-abc123.tmp"); + // A child that has already exited: its process ID is not running. + const exitedPID = spawnSync(process.execPath, ["-e", ""]).pid; + const orphan = join(originals, `.quak-${exitedPID}-abc123.tmp`); + const orphanThumb = join(thumbnails, `.quak-${exitedPID}-abc456.tmp`); + // This test's own process stands in for another process still + // downloading into the same cache. + const inProgress = join(originals, `.quak-${process.pid}-def123.tmp`); + const inProgressThumb = join( + thumbnails, + `.quak-${process.pid}-def456.tmp`, + ); const complete = join(originals, "1.jpg"); const thumb = join(thumbnails, "2.jpg"); writeFileSync(orphan, "half-written"); + writeFileSync(orphanThumb, "half-written"); + writeFileSync(inProgress, "half-written"); + writeFileSync(inProgressThumb, "half-written"); writeFileSync(complete, "whole"); writeFileSync(thumb, "whole-thumb"); @@ -177,8 +191,12 @@ describe("ContentCache.open", () => { await cache.open(); expect(existsSync(orphan)).toBe(false); + expect(existsSync(orphanThumb)).toBe(false); + expect(existsSync(inProgress)).toBe(true); + expect(existsSync(inProgressThumb)).toBe(true); expect(existsSync(complete)).toBe(true); expect(existsSync(thumb)).toBe(true); + expect(cache.pathsFor(1)).toEqual({ originalPath: complete }); }); it("records already-cached files so their paths appear in pathsFor", async () => { @@ -226,6 +244,25 @@ describe("ContentCache.original / thumbnail", () => { expect(skips).toEqual(["skipped"]); }); + it("takes only a letters-and-digits extension from the title", async () => { + // The title comes from the server; an extension such as `.\..\x` + // must not reach the cache file name, so it becomes `.bin`. + const { cache } = buildCache({ + files: [file(1, "a.jpg"), file(2, "b.\\..\\x"), file(3, "")], + }); + await cache.open(); + + expect((await cache.original(1)).path).toBe( + join(cacheDir, "originals", "1.jpg"), + ); + expect((await cache.original(2)).path).toBe( + join(cacheDir, "originals", "2.bin"), + ); + expect((await cache.original(3)).path).toBe( + join(cacheDir, "originals", "3.bin"), + ); + }); + it("serves a file already present in the download directory without fetching", async () => { const downloadDirectory = join(root, "backup"); mkdirSync(join(downloadDirectory, "originals"), { recursive: true }); diff --git a/test/library/fresh.test.ts b/test/library/fresh.test.ts index 15b71b7..10b8ee4 100644 --- a/test/library/fresh.test.ts +++ b/test/library/fresh.test.ts @@ -179,7 +179,7 @@ describe("Library.fresh", () => { // And the change is now live for the default namespaces too. expect(lib.photos.byID({ fileID: 1002 })?.fileID).toBe(1002); } finally { - lib.close(); + await lib.close(); } }); @@ -228,7 +228,7 @@ describe("Library.fresh", () => { expect(reads.photos.byID({ fileID: 1002 })?.fileID).toBe(1002); } } finally { - lib.close(); + await lib.close(); } }); @@ -266,7 +266,7 @@ describe("Library.fresh", () => { ); expect(lib.status().lastError).toBeUndefined(); } finally { - lib.close(); + await lib.close(); } }); }); diff --git a/test/library/library.test.ts b/test/library/library.test.ts index fe951c3..2669871 100644 --- a/test/library/library.test.ts +++ b/test/library/library.test.ts @@ -17,7 +17,8 @@ * failure surfaces via `onProgress` ("failed") and `status()`, and a later * success clears the error. `open()` itself resolves even when the first * refresh fails (offline start from cache). - * 5. `close()` stops the timer and is idempotent. + * 5. `close()` stops the timer and is idempotent, and its promise resolves + * only once an in-flight refresh has written the cache file. * 6. `cacheDirectory` defaults to the env-paths cache dir plus the user id. * 7. `open()` branches on the cache: an empty cache awaits the first refresh * (it has nothing to serve yet); an existing cache serves its copy at once @@ -37,6 +38,11 @@ * short interval and `vi.waitFor`: a fake clock cannot settle the real * fsync-and-rename cache write, and empty diffs never write, so the eventual * state is stable to poll for. + * + * A refresh changes RAM before it writes the cache file, so a polled state can + * be visible while that write is still running. Every test therefore awaits + * `close()`, which waits for the in-flight refresh, before `afterEach` removes + * the directory. */ import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; @@ -193,7 +199,7 @@ describe("Library.open and background refresh", () => { expect(reloaded.getFile(1, 1001)?.id).toBe(1001); expect(reloaded.collectionsSinceTime).toBe(100); } finally { - lib.close(); + await lib.close(); } }); @@ -224,7 +230,7 @@ describe("Library.open and background refresh", () => { expect(client.collectionsSinceTimes.length).toBe(collectionCalls); expect(client.filesCalls.length).toBe(fileCalls); } finally { - lib.close(); + await lib.close(); } }); @@ -271,7 +277,7 @@ describe("Library.open and background refresh", () => { { timeout: 2000, interval: 5 }, ); } finally { - lib.close(); + await lib.close(); } }); @@ -303,7 +309,7 @@ describe("Library.open and background refresh", () => { // never re-fetched. expect(client.filesCalls).toEqual([]); } finally { - lib.close(); + await lib.close(); } }); @@ -357,7 +363,7 @@ describe("Library.open and background refresh", () => { { timeout: 2000, interval: 5 }, ); } finally { - lib.close(); + await lib.close(); } }); @@ -402,7 +408,7 @@ describe("Library.open and background refresh", () => { await new Promise((r) => setTimeout(r, FAST_INTERVAL * 1000 * 4)); expect(saveSpy).toHaveBeenCalledTimes(2); } finally { - lib.close(); + await lib.close(); saveSpy.mockRestore(); } }); @@ -462,7 +468,7 @@ describe("Library.open and background refresh", () => { { timeout: 2000, interval: 5 }, ); } finally { - lib.close(); + await lib.close(); } }); @@ -488,7 +494,7 @@ describe("Library.open and background refresh", () => { ), ).toBe(true); } finally { - lib.close(); + await lib.close(); } }); @@ -502,9 +508,14 @@ describe("Library.open and background refresh", () => { seed.putFile(file(1001, 1, 400)); await seed.save(); - // The server never answers this run's first refresh. + // The server does not answer this run's first refresh until the test + // is done with it. + let answerFirstFetch: (page: CollectionsPage) => void = () => {}; const client = new MockClient(); - client.collectionsSince = () => new Promise<CollectionsPage>(() => {}); + client.collectionsSince = () => + new Promise<CollectionsPage>((resolve) => { + answerFirstFetch = resolve; + }); // open() must resolve from the cache without blocking on the network, // and reads must serve the seeded copy. @@ -517,7 +528,9 @@ describe("Library.open and background refresh", () => { expect(lib.status().lastRefreshAt).toBeUndefined(); expect(lib.status().lastError).toBeUndefined(); } finally { - lib.close(); + // close() waits for the outstanding refresh, so let it finish. + answerFirstFetch({ collections: [], deleted: [], cursor: 500 }); + await lib.close(); } }); @@ -564,7 +577,7 @@ describe("Library.open and background refresh", () => { expect(lib.listFiles(1).map((f) => f.id)).toEqual([1001]); expect(lib.status().lastRefreshAt).toBeGreaterThan(0); } finally { - lib.close(); + await lib.close(); } }); @@ -619,7 +632,7 @@ describe("Library.open and background refresh", () => { ); expect(reloaded.getFile(1, 1001)?.id).toBe(1001); } finally { - lib.close(); + await lib.close(); saveSpy.mockRestore(); } }); @@ -634,8 +647,8 @@ describe("Library.open and background refresh", () => { }); const callsAfterOpen = client.collectionsSinceTimes.length; - lib.close(); - lib.close(); // second close must not throw + await lib.close(); + await lib.close(); // second close must not throw expect(lib.status().closed).toBe(true); // No further refreshes fire once closed. @@ -643,6 +656,65 @@ describe("Library.open and background refresh", () => { expect(client.collectionsSinceTimes.length).toBe(callsAfterOpen); }); + it("close() resolves only after an in-flight refresh has written the cache", async () => { + const path = join(cacheDirectory, "metadata.json"); + const seed = await MetadataStore.load(path); + seed.userID = USER_ID; + seed.collectionsSinceTime = 500; + seed.putCollection(collection(1, 400)); + seed.putFile(file(1001, 1, 400)); + await seed.save(); + + const client = new MockClient(); + client.collectionsQueue.push({ + collections: [collection(1, 600)], + deleted: [], + cursor: 600, + }); + client.filesFor(1, { + files: [file(1002, 1, 600)], + deleted: [], + cursor: 600, + }); + + // Hold the refresh's cache write until the test releases it. + const realSave = MetadataStore.prototype.save; + let releaseSave: () => void = () => {}; + const saveHeld = new Promise<void>((resolve) => { + releaseSave = resolve; + }); + const saveSpy = vi + .spyOn(MetadataStore.prototype, "save") + .mockImplementation(async function (this: MetadataStore) { + await saveHeld; + return realSave.call(this); + }); + + const lib = await Library.open({ client, cacheDirectory }); + try { + await vi.waitFor(() => expect(saveSpy).toHaveBeenCalled(), { + timeout: 2000, + interval: 5, + }); + + let closed = false; + const closing = lib.close().then(() => { + closed = true; + }); + await new Promise((r) => setTimeout(r, 50)); + expect(closed).toBe(false); + + releaseSave(); + await closing; + const reloaded = await MetadataStore.load(path); + expect(reloaded.getFile(1, 1002)?.id).toBe(1002); + } finally { + releaseSave(); + await lib.close(); + saveSpy.mockRestore(); + } + }); + it("defaults cacheDirectory to the env-paths cache dir plus user id", async () => { const xdg = join(dir, "xdg-cache"); const prev = process.env.XDG_CACHE_HOME; @@ -659,7 +731,7 @@ describe("Library.open and background refresh", () => { expect(lib.cacheDirectory.startsWith(xdg)).toBe(true); expect(lib.cacheDirectory.endsWith(String(USER_ID))).toBe(true); } finally { - lib.close(); + await lib.close(); } } finally { if (prev === undefined) delete process.env.XDG_CACHE_HOME; diff --git a/test/library/mldata.test.ts b/test/library/mldata.test.ts index 6022db7..3de1cdc 100644 --- a/test/library/mldata.test.ts +++ b/test/library/mldata.test.ts @@ -13,6 +13,8 @@ * 2. `Library` wiring: after each refresh the library fetches ML data through * the metadata pool for every known file not yet cached, is incremental on * later refreshes, and refetches a file whose `updationTime` advanced. + * Opening a cache directory written by another account starts empty, + * its ML data included (issue #104). * * Embedding values are chosen to be exactly representable as float32 so the * round-trip through `clip.f32` compares equal. @@ -374,7 +376,7 @@ describe("Library ML-data fetch on refresh", () => { await new Promise((r) => setTimeout(r, FAST_INTERVAL * 1000 * 5)); expect(client.mlFetchCalls.length).toBe(callsAfterFirst); } finally { - lib.close(); + await lib.close(); } }); @@ -443,7 +445,118 @@ describe("Library ML-data fetch on refresh", () => { expect(client.mlFetchCalls.length).toBeGreaterThan(callsBefore); expect(client.mlFetchCalls.flat()).toContain(1001); } finally { - lib.close(); + await lib.close(); + } + }); + + it("close() resolves only after a running ML data fetch has stored its payloads", async () => { + const client = new MLMockClient(); + client.collectionsQueue.push({ + collections: [collection(1, 100)], + deleted: [], + cursor: 100, + }); + client.filesFor(1, { + files: [file(1001, 1, 90)], + deleted: [], + cursor: 90, + }); + client.mlByFile.set(1001, payload([0.5, 0.25, 0.75])); + + // Hold the ML data fetch open until the test releases it. + let release!: () => void; + const held = new Promise<void>((r) => (release = r)); + let fetchStarted!: () => void; + const started = new Promise<void>((r) => (fetchStarted = r)); + const realFetch = client.fetchMLData.bind(client); + client.fetchMLData = async (args) => { + fetchStarted(); + await held; + return realFetch(args); + }; + + const lib = await Library.open({ + client, + cacheDirectory, + refreshIntervalSeconds: 3600, + }); + try { + await started; + + let closed = false; + const closing = lib.close().then(() => { + closed = true; + }); + await new Promise((r) => setTimeout(r, 20)); + expect(closed).toBe(false); + + release(); + await closing; + expect( + existsSync(join(cacheDirectory, "mldata", "1001.json")), + ).toBe(true); + } finally { + release(); + await lib.close(); + } + }); + + it("starts empty when the cache directory holds another account's cache", async () => { + // Account A fills the cache directory: metadata and ML data. + const clientA = new MLMockClient(); + clientA.collectionsQueue.push({ + collections: [collection(1, 100)], + deleted: [], + cursor: 100, + }); + clientA.filesFor(1, { + files: [file(1001, 1, 90)], + deleted: [], + cursor: 90, + }); + clientA.mlByFile.set(1001, payload([0.5, 0.25, 0.75])); + const libA = await Library.open({ + client: clientA, + cacheDirectory, + refreshIntervalSeconds: 3600, + }); + try { + await vi.waitFor( + () => expect(libA.status().lastMLFetchAt).toBeGreaterThan(0), + { timeout: 2000, interval: 5 }, + ); + } finally { + await libA.close(); + } + + // Account B opens the same directory. + const clientB = new MLMockClient(); + clientB.userID = USER_ID + 1; + const sinceTimes: number[] = []; + const realCollectionsSince = clientB.collectionsSince.bind(clientB); + clientB.collectionsSince = async (args) => { + sinceTimes.push(args.sinceTime); + return realCollectionsSince(args); + }; + const libB = await Library.open({ + client: clientB, + cacheDirectory, + refreshIntervalSeconds: 3600, + }); + try { + expect(sinceTimes[0]).toBe(0); + expect(libB.status().userID).toBe(USER_ID + 1); + expect(libB.listCollections()).toEqual([]); + expect(libB.getFile(1, 1001)).toBeUndefined(); + expect(await libB.mldata.forFile({ fileID: 1001 })).toBeUndefined(); + expect( + libB.mldata.searchByEmbedding({ embedding: [0.5, 0.25, 0.75] }), + ).toEqual([]); + expect( + existsSync(join(cacheDirectory, "mldata", "1001.json")), + ).toBe(false); + } finally { + await libB.close(); } }); }); diff --git a/test/library/precache.test.ts b/test/library/precache.test.ts index 2580579..977867f 100644 --- a/test/library/precache.test.ts +++ b/test/library/precache.test.ts @@ -402,35 +402,97 @@ describe("Precache through Library.open", () => { } it("starts both precaches from open() and reports them in status()", async () => { - const thumbFetched = new Set<number>(); - const origFetched = new Set<number>(); const source: ContentSource = { - original: async ({ file: f, destination }) => { - origFetched.add(f.id); + original: async ({ destination }) => { await writeFile(destination, Buffer.alloc(10, 1)); return { bytesWritten: 10 }; }, - thumbnail: async ({ file: f, destination }) => { - thumbFetched.add(f.id); + thumbnail: async ({ destination }) => { await writeFile(destination, Buffer.alloc(10, 1)); return { bytesWritten: 10 }; }, }; + // Each fill reports "done" once the cache has recorded its files. The + // source returning is not enough: the cache records a file only after + // it has checked it on disk. + const finished = new Set<string>(); + let bothFinished!: () => void; + const precached = new Promise<void>((r) => (bothFinished = r)); const lib = await Library.open({ client: new MockClient(), cacheDirectory: join(root, "cache"), contentSource: source, refreshIntervalSeconds: 3600, + onProgress: (e) => { + if ( + e.status === "done" && + (e.operation === "precacheThumbnails" || + e.operation === "precacheOriginals") + ) { + finished.add(e.operation); + if (finished.size === 2) bothFinished(); + } + }, }); // Every file's thumbnail is precached; the favorite (file 3) and the // week's files (1, 2) all have their originals precached. - await until(() => thumbFetched.size === 3 && origFetched.size === 3); + await precached; const status = lib.status(); expect(status.thumbnailsTotal).toBe(3); expect(status.thumbnailsCached).toBe(3); expect(status.originalsPinned).toBe(3); expect(status.originalsCached).toBe(3); - lib.close(); + await lib.close(); }); + + it.each(["thumbnail", "original"] as const)( + "close() resolves only after a running %s precache fetch has written its file", + async (kind) => { + // Only the fill under test runs, and each of its fetches waits + // until the test releases it. + let release!: () => void; + const held = new Promise<void>((r) => (release = r)); + let fetchStarted!: (destination: string) => void; + const started = new Promise<string>((r) => (fetchStarted = r)); + const fetch = async ({ destination }: { destination: string }) => { + fetchStarted(destination); + await held; + await writeFile(destination, Buffer.alloc(10, 1)); + return { bytesWritten: 10 }; + }; + const unused = async () => { + throw new Error("this fill is turned off"); + }; + const source: ContentSource = + kind === "thumbnail" + ? { original: unused, thumbnail: fetch } + : { original: fetch, thumbnail: unused }; + const lib = await Library.open({ + client: new MockClient(), + cacheDirectory: join(root, "cache"), + contentSource: source, + refreshIntervalSeconds: 3600, + precacheThumbnails: kind === "thumbnail", + precacheOriginals: kind === "original", + }); + try { + const destination = await started; + + let closed = false; + const closing = lib.close().then(() => { + closed = true; + }); + await new Promise((r) => setTimeout(r, 20)); + expect(closed).toBe(false); + + release(); + await closing; + expect(existsSync(destination)).toBe(true); + } finally { + release(); + await lib.close(); + } + }, + ); }); diff --git a/test/library/read.test.ts b/test/library/read.test.ts index 102bdee..2fe394a 100644 --- a/test/library/read.test.ts +++ b/test/library/read.test.ts @@ -531,7 +531,7 @@ describe("Library exposes the read surface over its live store", () => { expect(client.collectionsCalls).toBe(collectionsBefore); expect(client.filesCalls).toBe(filesBefore); } finally { - lib.close(); + await lib.close(); } }); }); diff --git a/test/library/snapshot.test.ts b/test/library/snapshot.test.ts index 511b4c4..c0f306c 100644 --- a/test/library/snapshot.test.ts +++ b/test/library/snapshot.test.ts @@ -159,7 +159,7 @@ describe("Library.snapshot and Library.subscribe", () => { for (const p of snap.photos) expect("key" in p).toBe(false); for (const a of snap.albums) expect("key" in a).toBe(false); } finally { - lib.close(); + await lib.close(); } }); @@ -219,7 +219,7 @@ describe("Library.snapshot and Library.subscribe", () => { expect(change.refreshedAt).toBeGreaterThan(0); } finally { unsubscribe(); - lib.close(); + await lib.close(); } }); @@ -251,7 +251,7 @@ describe("Library.snapshot and Library.subscribe", () => { expect(changes).toEqual([]); } finally { unsubscribe(); - lib.close(); + await lib.close(); } }); @@ -297,7 +297,7 @@ describe("Library.snapshot and Library.subscribe", () => { ); expect(changes).toEqual([]); } finally { - lib.close(); + await lib.close(); } }); }); diff --git a/test/model/decrypt.test.ts b/test/model/decrypt.test.ts index 049f92c..f0db636 100644 --- a/test/model/decrypt.test.ts +++ b/test/model/decrypt.test.ts @@ -146,10 +146,13 @@ const buildSharedRawCollection = ( const buildRawFile = ( collectionKey: Uint8Array, opts?: { - title?: string; + // Any JSON value; `undefined` leaves the title out of the metadata. + title?: unknown; fileType?: number; creationTime?: number; info?: { fileSize?: number; thumbSize?: number }; + // Replaces the whole metadata JSON value. + metadata?: unknown; }, ): RawEnteFile => { const fileKey = sodium.crypto_secretbox_keygen(); @@ -158,8 +161,8 @@ const buildRawFile = ( collectionKey, ); - const metadata = { - title: opts?.title ?? "IMG_0001.jpg", + const defaultMetadata = { + title: opts && "title" in opts ? opts.title : "IMG_0001.jpg", fileType: opts?.fileType ?? 0, creationTime: opts?.creationTime ?? 1700000000000000, modificationTime: 1700000000000000, @@ -167,6 +170,8 @@ const buildRawFile = ( longitude: 2.3522, hash: "abcdef1234567890", }; + const metadata = + opts && "metadata" in opts ? opts.metadata : defaultMetadata; // File metadata is encrypted as a single-chunk secretstream blob // (not secretbox). The decryptionHeader is the secretstream init header. const metadataBytes = new TextEncoder().encode(JSON.stringify(metadata)); @@ -321,6 +326,71 @@ describe("model.decryptFile", () => { expect(file.metadata.longitude).toBeCloseTo(2.3522); }); + it("reads a missing or non-string title as an empty string", () => { + // The server controls the metadata JSON. A title that is not a + // string must not reach code that builds file names from it. + const masterKey = sodium.crypto_secretbox_keygen(); + const { collectionKey } = buildRawCollection(masterKey); + for (const title of [undefined, null, 42, ["a"], { x: "../y" }]) { + const file = decryptFile( + buildRawFile(collectionKey, { title }), + collectionKey, + ); + expect(file.metadata.title).toBe(""); + } + }); + + it("rejects metadata that is not a JSON object", () => { + const masterKey = sodium.crypto_secretbox_keygen(); + const { collectionKey } = buildRawCollection(masterKey); + for (const metadata of [null, "IMG_0001.jpg", 7, []]) { + const raw = buildRawFile(collectionKey, { metadata }); + expect(() => decryptFile(raw, collectionKey)).toThrow( + "file 200: metadata is not a JSON object", + ); + } + }); + + it("reads the recorded content hash, joining an older live photo's two parts", () => { + const masterKey = sodium.crypto_secretbox_keygen(); + const { collectionKey } = buildRawCollection(masterKey); + const hashOf = (metadata: Record<string, unknown>) => + decryptFile( + buildRawFile(collectionKey, { + metadata: { title: "x", ...metadata }, + }), + collectionKey, + ).metadata.hash; + + expect(hashOf({ fileType: 0, hash: "H" })).toBe("H"); + expect( + hashOf({ fileType: 2, hash: "H", imageHash: "I", videoHash: "V" }), + ).toBe("H"); + expect(hashOf({ fileType: 2, imageHash: "I", videoHash: "V" })).toBe( + "I:V", + ); + expect(hashOf({ fileType: 2, imageHash: "I" })).toBeUndefined(); + expect( + hashOf({ fileType: 0, imageHash: "I", videoHash: "V" }), + ).toBeUndefined(); + expect(hashOf({ fileType: 0 })).toBeUndefined(); + expect(hashOf({ fileType: 0, hash: 42 })).toBeUndefined(); + expect( + hashOf({ fileType: 2, imageHash: "I", videoHash: 7 }), + ).toBeUndefined(); + // An empty string counts as absent, not as a hash to match. + expect(hashOf({ fileType: 0, hash: "" })).toBeUndefined(); + expect( + hashOf({ fileType: 2, hash: "", imageHash: "I", videoHash: "V" }), + ).toBe("I:V"); + expect( + hashOf({ fileType: 2, imageHash: "", videoHash: "V" }), + ).toBeUndefined(); + expect( + hashOf({ fileType: 2, imageHash: "I", videoHash: "" }), + ).toBeUndefined(); + }); + it("maps fileType numbers to FileType strings", () => { // Ente uses: 0=image, 1=video, 2=livePhoto const masterKey = sodium.crypto_secretbox_keygen(); diff --git a/test/packaging/build-context.test.ts b/test/packaging/build-context.test.ts index 056fb57..d6c24e4 100644 --- a/test/packaging/build-context.test.ts +++ b/test/packaging/build-context.test.ts @@ -2,13 +2,13 @@ // failures are silent. // // Excluding too little: a worktree left under `.claude/` is copied into the -// image, vitest globs its `test/` tree as well as the real one, and the -// containerised `make check` runs the whole suite twice over while reporting -// success. A compiled `bin/quak` is ~100 MB of context nobody needs. +// image, vitest globs its `test/` tree as well as the real one, and the test +// phase runs the whole suite twice over while reporting success. A compiled +// `bin/quak` is ~100 MB of context nobody needs. // // Excluding too much: Prettier 3 reads `.gitignore` as a default ignore file, -// so dropping it from the context silently changes which files -// `make fmt-check` looks at inside the image compared to the host. +// so dropping it from the context silently changes which files the lint +// phase's prettier check looks at compared to `make fmt-check` on the host. // // Neither shows up as a build failure, so they are asserted here. import { describe, expect, it } from "vitest"; @@ -47,18 +47,13 @@ describe(".dockerignore", () => { expect(dockerignore).not.toContain(".gitignore"); }); - // Both images are built from this same context, and the lint image runs - // eslint and prettier across it. BuildKit lets a `<dockerfile>.dockerignore` - // shadow the root one for a single build; such a file would silently give - // the lint build a different, unreviewed context — and eslint's flat config - // does not ignore dot-directories, so a stray `.claude/` worktree would be - // linted. - it.each(["Dockerfile", "Dockerfile.lint"])( - "is not shadowed by a per-Dockerfile ignore file for %s", - (name) => { - expect(existsSync(join(repoRoot, `${name}.dockerignore`))).toBe( - false, - ); - }, - ); + // BuildKit lets a `Dockerfile.dockerignore` shadow the root one; such a + // file would silently give the build a different, unreviewed context — + // and eslint's flat config does not ignore dot-directories, so a stray + // `.claude/` worktree would be linted. + it("is not shadowed by a Dockerfile.dockerignore", () => { + expect(existsSync(join(repoRoot, "Dockerfile.dockerignore"))).toBe( + false, + ); + }); }); diff --git a/test/packaging/entrypoints.test.ts b/test/packaging/entrypoints.test.ts index 58149c5..a9b5426 100644 --- a/test/packaging/entrypoints.test.ts +++ b/test/packaging/entrypoints.test.ts @@ -1,9 +1,7 @@ // The package manifest promises three files that only exist after a build: // `main`, `types`, and the `quak` binary. Nothing in the test suite used to -// look at them, and `make check` runs the suite and the lint container but -// never the build, so `tsconfig.json` and `package.json` were free to drift -// apart. (The formatting check is part of the lint container, not a step of -// its own; `test/packaging/lint-once.test.ts` is what holds that shape.) They +// look at them, and `make check` runs the test and lint phases but never the +// build, so `tsconfig.json` and `package.json` were free to drift apart. They // did: `rootDir` was `./src` while `include` also pulled in `bin/**/*`, which // is TS6059, and no build had succeeded for as long as that was true. // diff --git a/test/packaging/lint-docker.test.ts b/test/packaging/lint-docker.test.ts deleted file mode 100644 index fa561f2..0000000 --- a/test/packaging/lint-docker.test.ts +++ /dev/null @@ -1,184 +0,0 @@ -// Linting runs in Docker, one way, everywhere: `script/lint` builds -// `Dockerfile.lint`, which COPYs the repo into a digest-pinned image and runs -// eslint and prettier as build steps, so a successful build IS a clean lint. -// -// Three things can quietly undo that, and none of them shows up as a build -// failure, which is why they are asserted here: -// -// 1. Recursion. `script/check` calls `script/lint`, and `script/lint` is now a -// `docker build`. Anything that runs `make check` inside a container is -// therefore asking for Docker inside Docker, and CI breaks. The image built -// from `Dockerfile` runs the suite and the compile only; lint happens once, -// in `Dockerfile.lint`. -// 2. Cache. A lint build over an unchanged tree returns success in well under a -// second having linted nothing. The `LINT_EPOCH` guard is what forces the -// linter layers to execute, and it has to fail closed: an unset build -// argument is the empty string, which is a perfectly stable cache key, so an -// invocation that omits it must be rejected rather than served a cached -// green. -// 3. A host lint path surviving alongside the container one, which would let a -// lint result come from an unpinned local toolchain. -import { describe, expect, it } from "vitest"; -import { readFileSync } from "node:fs"; -import { fileURLToPath } from "node:url"; -import { join } from "node:path"; - -const repoRoot = fileURLToPath(new URL("../../", import.meta.url)); - -const read = (name: string): string => - readFileSync(join(repoRoot, name), "utf-8"); - -// The executable lines of a shell script or Dockerfile: comments carry the -// reasoning and frequently name the very commands these tests forbid, so they -// would otherwise trigger every assertion below. -const instructions = (name: string): string[] => - read(name) - .split("\n") - .map((line) => line.trim()) - .filter((line) => line !== "" && !line.startsWith("#")); - -const lintScript = instructions("script/lint"); -const dockerfileLint = instructions("Dockerfile.lint"); -const dockerfile = instructions("Dockerfile"); -const cibuild = instructions("script/cibuild"); - -const has = (lines: string[], pattern: RegExp): boolean => - lines.some((line) => pattern.test(line)); - -describe("script/lint", () => { - it("lints by building Dockerfile.lint", () => { - expect(has(lintScript, /docker build .*-f Dockerfile\.lint/)).toBe( - true, - ); - }); - - // The whole point of the ruling: no invocation of a linter against the - // working tree survives, so a lint verdict can only come from the pinned - // image. - it("runs no linter on the host", () => { - expect(has(lintScript, /eslint|prettier/)).toBe(false); - }); - - // Without a fresh epoch the build is served from cache in under a second, - // having linted nothing, and still exits 0. - it("passes a fresh LINT_EPOCH on every run", () => { - expect( - has(lintScript, /--build-arg LINT_EPOCH="\$\(date \+%s\)"/), - ).toBe(true); - }); -}); - -describe("Dockerfile.lint", () => { - // Tag references are server-mutable, so they are remote code execution. - it("pins its base image by digest", () => { - expect(has(dockerfileLint, /^FROM \S+@sha256:[0-9a-f]{64}/)).toBe(true); - }); - - it("runs eslint as a build step", () => { - expect(has(dockerfileLint, /^RUN .*eslint \./)).toBe(true); - }); - - it("runs prettier as a build step", () => { - expect(has(dockerfileLint, /^RUN .*prettier --check \./)).toBe(true); - }); - - // An unset ARG is the empty string, and an empty string is a perfectly - // stable cache key. Rejecting it is what stops a bare - // `docker build -f Dockerfile.lint .` from reporting a green it did not - // earn. - it("refuses to build without LINT_EPOCH", () => { - expect(has(dockerfileLint, /^ARG LINT_EPOCH$/)).toBe(true); - expect( - has(dockerfileLint, /^RUN \[ -n "\$LINT_EPOCH" \] \|\| exit 1$/), - ).toBe(true); - }); - - // The guard only forces execution of the layers below it, so both linters - // have to sit after it. Layer order is the mechanism, not a style choice. - it("puts both linters below the epoch guard", () => { - const guard = dockerfileLint.findIndex((line) => - /^RUN \[ -n "\$LINT_EPOCH" \]/.test(line), - ); - const linters = dockerfileLint - .map((line, index) => ({ line, index })) - .filter(({ line }) => /^RUN .*(eslint|prettier)/.test(line)); - - expect(linters.length).toBeGreaterThan(0); - for (const { line, index } of linters) { - expect( - index, - `${line} must run below the LINT_EPOCH guard`, - ).toBeGreaterThan(guard); - } - }); - - // Dependency installation is the slow layer and has nothing to do with the - // sources, so it caches separately: manifests first, sources afterwards. - it("copies the manifests before the sources", () => { - const manifests = dockerfileLint.findIndex((line) => - /^COPY package\.json yarn\.lock/.test(line), - ); - const sources = dockerfileLint.findIndex((line) => - /^COPY \. \.$/.test(line), - ); - - expect(manifests).toBeGreaterThanOrEqual(0); - expect(sources).toBeGreaterThan(manifests); - }); - - // script/lint is a docker build; a lint step that shelled out to it would - // recurse. - it("does not call script/lint or make lint", () => { - expect(has(dockerfileLint, /make lint|script\/lint/)).toBe(false); - }); -}); - -describe("Dockerfile", () => { - // `make check` runs script/lint, which is a docker build, so an image that - // ran it would need a Docker daemon inside the container. - it("does not run make check, make lint or script/lint", () => { - expect( - has(dockerfile, /make check|make lint|script\/(check|lint)/), - ).toBe(false); - }); - - // The replaced lint stage took a `COPY --from=lint` dependency to order - // itself before the check stage. Dockerfile.lint is that stage now, and - // two definitions of how to lint is one too many. - it("has no lint stage", () => { - expect(has(dockerfile, /AS lint\b|--from=lint\b/)).toBe(false); - }); - - it("still runs the suite and the build under the epoch guard", () => { - expect(has(dockerfile, /^RUN make test$/)).toBe(true); - expect(has(dockerfile, /^RUN make build$/)).toBe(true); - expect( - has(dockerfile, /^RUN \[ -n "\$CHECK_EPOCH" \] \|\| exit 1$/), - ).toBe(true); - }); -}); - -describe("script/cibuild", () => { - // CI has to get both verdicts. Lint goes first so the fast failure is - // reported before the suite runs. - it("builds the lint image before the test and build image", () => { - const lint = cibuild.findIndex((line) => /\/lint"/.test(line)); - const check = cibuild.findIndex((line) => - /docker build .*CHECK_EPOCH/.test(line), - ); - - expect(lint).toBeGreaterThanOrEqual(0); - expect(check).toBeGreaterThan(lint); - }); -}); - -describe("package.json", () => { - // `yarn lint` was a second, unpinned way to get a lint verdict, from - // whatever eslint the working tree happened to have installed. - it("exposes no host lint script", () => { - const pkg = JSON.parse(read("package.json")) as { - scripts: Record<string, string>; - }; - expect(pkg.scripts.lint).toBeUndefined(); - }); -}); diff --git a/test/packaging/lint-once.test.ts b/test/packaging/lint-once.test.ts deleted file mode 100644 index 16fbad1..0000000 --- a/test/packaging/lint-once.test.ts +++ /dev/null @@ -1,503 +0,0 @@ -// `make check` used to run `prettier --check .` twice: once inside the lint -// container (`script/lint` builds `Dockerfile.lint`, which runs eslint and -// prettier as build steps) and once again on the host, because `script/check` -// also called `script/fmt-check`. Two passes, one verdict, and the host one is -// the weaker of the two — its prettier is whatever the working tree happens to -// have installed, while the container's is digest-pinned and installed under -// `--frozen-lockfile`. -// -// The fix was to delete the host call from `script/check` and `script/precommit`. -// Nothing about that fix is self-enforcing: anyone can wire `script/fmt-check` -// back in, or add a prettier step to a Dockerfile, and every build stays green -// while quietly doing the work twice again. So the count is asserted here -// rather than promised in a comment. -// -// The assertion is a static walk of the invocation graph, not a string match -// against one file. Starting from an entrypoint, it follows every edge the repo -// actually uses to reach another command — `run:` steps in the CI workflow, -// `"$SCRIPT_DIR/<name>"` and `script/<name>` into other scripts, `make <target>` -// through the Makefile shims, `yarn run <name>` through the `package.json` -// scripts, and `docker build -f <file>` into that Dockerfile's `RUN` steps — and -// counts the prettier invocations it finds. A prettier call added anywhere in -// that graph is therefore caught, wherever it is added. -// -// Two entrypoints are walked, because they cover different graphs: `make check` -// is what a developer runs, and `.gitea/workflows/check.yml` is what CI runs. -// The CI walk starts at the workflow file rather than at a hand-picked script, -// so "the path CI executes" is read out of the repo instead of assumed; it -// reaches `script/cibuild`, and through it the `Dockerfile` image that `make -// check` never touches. Walking only `make check` is how a duplicate prettier -// pass in `Dockerfile` stayed invisible. -// -// Undercounting is the failure mode that would make this test worthless. Three -// things guard against it: the walk is asserted to have reached the nodes that -// matter, an unresolvable or empty node is a thrown error rather than a quiet -// zero, and prettier is counted per occurrence rather than per line, so two -// invocations chained with `&&` cannot read as one. -import { describe, expect, it } from "vitest"; -import { readFileSync } from "node:fs"; -import { fileURLToPath } from "node:url"; -import { join } from "node:path"; - -const repoRoot = fileURLToPath(new URL("../../", import.meta.url)); - -const read = (name: string): string => - readFileSync(join(repoRoot, name), "utf-8"); - -// A backslash at end of line continues the command; the resolver has to see the -// whole invocation, since the interesting flags (`-f Dockerfile.lint`) can sit -// on the continuation. -const joinContinuations = (text: string): string[] => { - const joined: string[] = []; - for (const raw of text.split("\n")) { - const line = raw.trim(); - const previous = joined[joined.length - 1]; - if (previous !== undefined && previous.endsWith("\\")) { - joined[joined.length - 1] = - `${previous.slice(0, -1).trim()} ${line}`; - } else { - joined.push(line); - } - } - return joined; -}; - -// Comments are stripped everywhere. The headers of these scripts explain the -// duplication this test exists to prevent, and therefore name `prettier` and -// `script/fmt-check` repeatedly; counting them would make the test assert the -// prose instead of the behaviour. -const executable = (text: string): string[] => - joinContinuations(text).filter( - (line) => line !== "" && !line.startsWith("#"), - ); - -// Every occurrence, not "does this line mention prettier": a line that reads -// `yarn run prettier --check . && yarn run prettier --check src` is two passes -// over the same tree, which is exactly the bug this file exists to catch, and -// counting it as one would hide it. `.prettierrc` and `.prettierignore` are not -// invocations and do not match, because `\b` requires a non-word character -// after the name. -const countPrettier = (line: string): number => - (line.match(/\bprettier\b/g) ?? []).length; - -// Makefile targets are thin shims (`check:` / tab / `@script/check`), so a -// `make <target>` edge has to resolve through them to keep "per `make check`" -// meaning what it says. Recipe lines are the tab-indented ones. -const makeRecipes = (): Map<string, string[]> => { - const recipes = new Map<string, string[]>(); - let current: string | null = null; - for (const raw of read("Makefile").split("\n")) { - if (raw.startsWith("\t")) { - if (current !== null) { - recipes.get(current)?.push(raw.trim().replace(/^[@-]+/, "")); - } - continue; - } - const target = /^([a-z][a-z-]*)\s*:(?!=)/.exec(raw); - current = target === null ? null : target[1]; - if (current !== null && !recipes.has(current)) { - recipes.set(current, []); - } - } - return recipes; -}; - -const recipes = makeRecipes(); - -const packageScripts = (): Record<string, string> => { - const pkg = JSON.parse(read("package.json")) as { - scripts?: Record<string, string>; - }; - return pkg.scripts ?? {}; -}; - -const scripts = packageScripts(); - -// Node keys: `script/<name>`, `docker:<Dockerfile>`, `make:<target>`, -// `yarn:<package.json script>`, `workflow:<CI workflow file>`. -const resolve = (node: string): string[] => { - if (node.startsWith("script/")) return executable(read(node)); - if (node.startsWith("docker:")) { - return executable(read(node.slice("docker:".length))) - .filter((line) => line.startsWith("RUN ")) - .map((line) => line.slice("RUN ".length)); - } - // The `run:` steps of a workflow, in file order. `uses:` steps are actions, - // not commands, and have no edges into this repo's graph. A `run: |` block - // would resolve to the bare `|`, which reaches nothing and therefore fails - // the count rather than passing quietly. - if (node.startsWith("workflow:")) { - return executable(read(node.slice("workflow:".length))) - .filter((line) => /^-?\s*run:\s*\S/.test(line)) - .map((line) => line.replace(/^-?\s*run:\s*/, "")); - } - if (node.startsWith("make:")) { - const target = node.slice("make:".length); - const recipe = recipes.get(target); - // A renamed or deleted target must be a loud failure: silently walking - // an empty recipe would report zero prettier invocations, which reads - // like the tidiest possible result. - if (recipe === undefined) { - throw new Error(`no such Makefile target: ${target}`); - } - return recipe; - } - if (node.startsWith("yarn:")) { - const name = node.slice("yarn:".length); - const script = scripts[name]; - if (script === undefined) { - throw new Error(`no such package.json script: ${name}`); - } - return [script]; - } - throw new Error(`unresolvable node: ${node}`); -}; - -// Same reasoning as the missing-target error, applied to every node kind: a -// node that resolves to no commands contributes zero prettier invocations and -// zero edges, which is indistinguishable from a clean result. Fail instead. -const commandsOf = (node: string): string[] => { - const commands = resolve(node); - if (commands.length === 0) { - throw new Error(`node resolved to no commands: ${node}`); - } - return commands; -}; - -const edgesOf = (line: string): string[] => { - const edges: string[] = []; - - // `"$SCRIPT_DIR/lint"`, `"$ROOT/script/lint"` and a bare `script/lint` are - // all the same edge. - for (const match of line.matchAll( - /(?:\$SCRIPT_DIR|\$\{SCRIPT_DIR\}|script)\/([a-z][a-z-]*)/g, - )) { - edges.push(`script/${match[1]}`); - } - - // Only real targets: `pkg_install gnumake make make make` in - // script/bootstrap is a package name, not an invocation of this Makefile. - for (const match of line.matchAll(/\bmake\s+([a-z][a-z-]*)/g)) { - if (recipes.has(match[1] ?? "")) edges.push(`make:${match[1]}`); - } - - // Same rule for yarn: `yarn run prettier` is the linter itself (counted, - // not followed), `yarn run fmt-check` would be a package.json script that - // runs it indirectly. - for (const match of line.matchAll(/\byarn(?:\s+run)?\s+([a-z][a-z-]*)/g)) { - if ((match[1] ?? "") in scripts) edges.push(`yarn:${match[1]}`); - } - - // The container lint pass lives behind a `docker build`; without following - // it the count would miss the one invocation that is supposed to survive. - if (/\bdocker\s+build\b/.test(line)) { - const file = /\s-f\s+(\S+)/.exec(line); - edges.push(`docker:${file === null ? "Dockerfile" : file[1]}`); - } - - return edges; -}; - -interface Walk { - prettier: number; - reached: Set<string>; -} - -// Repeated invocations must count repeatedly — running the same script twice is -// exactly the bug — so nodes are not deduplicated. The path stack is only there -// to turn a cycle into a loud failure instead of a hang. -// -// Counting and edge-following both happen for every line: a line that invokes -// prettier can also invoke something else, and skipping the edges of counted -// lines silently truncated the graph. -const walk = (node: string, path: string[] = [], into?: Walk): Walk => { - const result = into ?? { prettier: 0, reached: new Set<string>() }; - if (path.includes(node)) { - throw new Error(`invocation cycle: ${[...path, node].join(" -> ")}`); - } - result.reached.add(node); - - for (const line of commandsOf(node)) { - result.prettier += countPrettier(line); - for (const edge of edgesOf(line)) { - walk(edge, [...path, node], result); - } - } - return result; -}; - -describe("prettier runs exactly once per make check", () => { - const check = walk("make:check"); - - // The headline assertion, and the one the issue is about. - it("invokes prettier once for the whole of make check", () => { - expect(check.prettier).toBe(1); - }); - - // Guards against the count being 1 (or 0) because the walk never got - // anywhere. `make check` has to reach the suite, the lint script, and the - // Dockerfile whose build IS the lint verdict. - it.each(["script/check", "script/test", "script/lint", "Dockerfile.lint"])( - "reaches %s while counting", - (node) => { - const key = node.startsWith("script/") ? node : `docker:${node}`; - expect([...check.reached]).toContain(key); - }, - ); - - // The one that survives is the container's, not the host's: that is the - // authoritative verdict, since a successful Dockerfile.lint build is what - // CI treats as proof of a clean tree. - it("keeps the surviving invocation inside the lint container", () => { - expect(walk("docker:Dockerfile.lint").prettier).toBe(1); - }); - - it("does not reach the host formatting check from make check", () => { - expect([...check.reached]).not.toContain("script/fmt-check"); - }); -}); - -describe("prettier runs exactly once per CI build", () => { - // Rooted at the workflow file, so this is the graph CI executes rather than - // the graph someone believed CI executes. `make check` cannot stand in for - // it: CI runs script/cibuild, which builds Dockerfile as well as - // Dockerfile.lint, and nothing under `make check` ever reads Dockerfile. - const ci = walk("workflow:.gitea/workflows/check.yml"); - - it("invokes prettier once for the whole CI build", () => { - expect(ci.prettier).toBe(1); - }); - - // script/cibuild is here because the workflow is asserted to run it; - // Dockerfile is here because it is the half of the CI graph that the - // `make check` walk cannot see. - it.each([ - "script/cibuild", - "script/lint", - "docker:Dockerfile.lint", - "docker:Dockerfile", - ])("reaches %s while counting", (node) => { - expect([...ci.reached]).toContain(node); - }); - - // The test and build image must not lint: linting is Dockerfile.lint's job, - // and a prettier step added here would be a second pass over the same tree - // for the same verdict — on the one path where it matters most. - it("keeps prettier out of the test and build image", () => { - expect(walk("docker:Dockerfile").prettier).toBe(0); - }); -}); - -describe("the standalone entrypoints still do what their names say", () => { - // REPO_POLICIES.md requires both `make lint` and `make fmt-check` to exist - // and mean something. Dropping fmt-check from script/check must not turn it - // into a target nobody can use, and must not leave `make check` passing - // because both halves became no-ops. - it("still checks formatting under make fmt-check", () => { - expect(walk("make:fmt-check").prettier).toBe(1); - }); - - it("still checks formatting under make lint", () => { - expect(walk("make:lint").prettier).toBe(1); - }); -}); - -describe("script/precommit", () => { - // Same duplication as script/check, same fix. The hook still catches a - // badly formatted tree before the commit lands, because script/lint is the - // container prettier run — that is the whole reason the host call could go. - it("checks formatting exactly once", () => { - expect(walk("script/precommit").prettier).toBe(1); - }); - - it("gets that check from the lint container", () => { - expect([...walk("script/precommit").reached]).toContain( - "docker:Dockerfile.lint", - ); - }); -}); - -// script/bootstrap installs the dependencies, and it has two install sites: one -// for the case where yarn has to be reached through nvm, and one for the case -// where yarn is already on PATH. A substring check against the whole file -// cannot tell them apart, so it reports the first and says nothing about the -// second — which is the one the containers take, because the pinned node image -// ships yarn. Both are resolved separately here. -const installBranches = (): { withoutYarn: string[]; withYarn: string[] } => { - const lines = executable(read("script/bootstrap")); - const open = lines.findIndex((line) => - /^install_js_deps\s*\(\)/.test(line), - ); - if (open === -1) { - throw new Error("script/bootstrap: no install_js_deps function"); - } - const close = lines.indexOf("}", open); - const body = lines.slice(open + 1, close === -1 ? undefined : close); - const guard = body.findIndex((line) => - /^if\b.*\bmissing yarn\b/.test(line), - ); - const otherwise = body.indexOf("else", guard); - const end = body.indexOf("fi", otherwise); - if (guard === -1 || otherwise === -1 || end === -1) { - throw new Error( - "script/bootstrap: install_js_deps is not the expected " + - "if missing yarn / else / fi shape", - ); - } - return { - withoutYarn: body.slice(guard + 1, otherwise), - withYarn: body.slice(otherwise + 1, end), - }; -}; - -// Every `yarn install` in the given lines, with its flags, so an unpinned -// install cannot hide next to a pinned one. -const yarnInstalls = (lines: string[]): string[] => - lines.flatMap((line) => - [...line.matchAll(/\byarn install\b[^"'&|;]*/g)].map((match) => - match[0].trim(), - ), - ); - -describe("host and container prettier cannot disagree", () => { - // With the host pass gone from `make check`, `make fmt-check` is the only - // host-side formatting check left, and the container is the gate. The two - // must keep producing the same verdict on the same tree, or a developer - // running `make fmt-check` gets a green that CI then rejects. - // - // Three things make them agree, and all three are load-bearing: - it("pins the same prettier for both", () => { - const pkg = JSON.parse(read("package.json")) as { - devDependencies: Record<string, string>; - }; - // An exact version, not a range: `^3.8.1` would let the container and - // the host resolve different builds with different formatting. - expect(pkg.devDependencies.prettier).toMatch(/^\d+\.\d+\.\d+$/); - }); - - it("installs from the lockfile on the branch the container takes", () => { - // Both images are FROM a node image, which ships yarn, so `missing - // yarn` is false and this is the branch that runs in the container. - const installs = yarnInstalls(installBranches().withYarn); - expect(installs).not.toHaveLength(0); - for (const install of installs) { - expect(install).toContain("--frozen-lockfile"); - } - }); - - it("installs from the lockfile on the nvm branch too", () => { - // Not the container's branch, but it is the one a developer without - // yarn on PATH gets, and their prettier has to match the container's. - const installs = yarnInstalls(installBranches().withoutYarn); - expect(installs).not.toHaveLength(0); - for (const install of installs) { - expect(install).toContain("--frozen-lockfile"); - } - }); - - it("runs script/bootstrap inside the lint container", () => { - // Without this the lockfile assertions above would be about a script - // the container never executes. - expect([...walk("docker:Dockerfile.lint").reached]).toContain( - "script/bootstrap", - ); - }); - - it("keeps .gitignore in the build context", () => { - // Prettier 3 reads .gitignore as a default ignore file, so excluding it - // from the context would change which files the container checks. - const dockerignore = read(".dockerignore") - .split("\n") - .map((line) => line.trim()); - expect(dockerignore).not.toContain(".gitignore"); - }); -}); - -describe("the walk cannot pass vacuously", () => { - // An earlier draft of this file computed a Makefile target as - // `node.slice("make:")` — a string where a number belongs, which coerces to - // NaN and made every target resolve to nothing. The count went to zero and - // an assertion of "not twice" would have been satisfied by a walk that had - // read nothing at all. Every way of reaching nothing is therefore an - // error here, and the ways are tested rather than assumed. - it("reports zero for a subgraph that does not run prettier", () => { - expect(walk("make:clean").prettier).toBe(0); - }); - - it("refuses a Makefile target that does not exist", () => { - expect(() => walk("make:no-such-target")).toThrow( - /no such Makefile target/, - ); - }); - - it("refuses a package.json script that does not exist", () => { - expect(() => walk("yarn:no-such-script")).toThrow( - /no such package.json script/, - ); - }); - - it("refuses a script that does not exist", () => { - expect(() => walk("script/no-such-script")).toThrow(/ENOENT/); - }); - - it("refuses a node that resolves to no commands", () => { - // .dockerignore has no RUN steps, standing in for a Dockerfile whose - // steps a restructure moved somewhere the resolver cannot see. - expect(() => walk("docker:.dockerignore")).toThrow( - /resolved to no commands/, - ); - }); - - it("refuses a node kind it does not understand", () => { - expect(() => walk("nonsense")).toThrow(/unresolvable node/); - }); - - it("refuses to walk in circles", () => { - expect(() => walk("make:check", ["script/check"])).toThrow( - /invocation cycle/, - ); - }); -}); - -describe("the resolver reads what the shell would run", () => { - // Counting per line is how `yarn run prettier --check . && yarn run - // prettier --check src` read as a single invocation. - it("counts every prettier invocation on a line", () => { - expect( - countPrettier( - "yarn run prettier --check . && yarn run prettier --check src", - ), - ).toBe(2); - }); - - it("does not count the config files as invocations", () => { - expect(countPrettier("COPY .prettierrc .prettierignore ./")).toBe(0); - }); - - // The counting `continue` also dropped every edge that shared a line with a - // prettier call, so a whole subtree could be hidden behind one `&&`. - it("still follows the edges of a line that invokes prettier", () => { - expect( - edgesOf('yarn run prettier --check . && "$SCRIPT_DIR/lint"'), - ).toContain("script/lint"); - }); - - it("resolves every spelling of a script call to one node", () => { - expect( - edgesOf('"$SCRIPT_DIR/lint" "${SCRIPT_DIR}/test" script/fmt'), - ).toEqual(["script/lint", "script/test", "script/fmt"]); - }); - - it("follows a bare docker build to Dockerfile and -f to its file", () => { - expect(edgesOf("docker build .")).toContain("docker:Dockerfile"); - expect(edgesOf("docker build -f Dockerfile.lint .")).toContain( - "docker:Dockerfile.lint", - ); - }); - - it("reads the run steps of the CI workflow and not its uses steps", () => { - expect(commandsOf("workflow:.gitea/workflows/check.yml")).toEqual([ - "script/cibuild", - ]); - }); -}); diff --git a/test/packaging/nested-checkout.test.ts b/test/packaging/nested-checkout.test.ts new file mode 100644 index 0000000..af26af6 --- /dev/null +++ b/test/packaging/nested-checkout.test.ts @@ -0,0 +1,50 @@ +// A checkout nested under `.claude/` must not add its tests to this suite. +// The test plants one in a temporary directory next to a real test file and +// asks vitest, with this repo's config, which test files it would run. +import { afterEach, describe, expect, it } from "vitest"; +import { execFileSync } from "node:child_process"; +import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; + +const repoRoot = fileURLToPath(new URL("../../", import.meta.url)); + +let root = ""; + +afterEach(() => { + rmSync(root, { recursive: true, force: true }); +}); + +const writeTest = (path: string): void => { + mkdirSync(join(root, path, ".."), { recursive: true }); + writeFileSync( + join(root, path), + 'import { it } from "vitest";\nit("runs", () => {});\n', + ); +}; + +describe("vitest.config.ts", () => { + it("does not collect tests from a checkout nested under .claude/", () => { + root = mkdtempSync(join(tmpdir(), "quak-nested-checkout-")); + writeTest("test/real.test.ts"); + writeTest(".claude/worktrees/other/test/real.test.ts"); + + const output = execFileSync( + process.execPath, + [ + join(repoRoot, "node_modules/vitest/vitest.mjs"), + "list", + "--filesOnly", + "--config", + join(repoRoot, "vitest.config.ts"), + "--root", + root, + ], + { cwd: root, encoding: "utf-8" }, + ); + + const files = output.split("\n").filter((line) => line !== ""); + expect(files).toEqual(["test/real.test.ts"]); + }); +}); diff --git a/test/retry/retry.test.ts b/test/retry/retry.test.ts index 7baf8c0..394d96e 100644 --- a/test/retry/retry.test.ts +++ b/test/retry/retry.test.ts @@ -149,8 +149,11 @@ describe("isRetryable: transport failures", () => { }); it("retries an errno carried on the error itself", () => { + // Every errno the classifier names, so none can be reclassified + // unnoticed. for (const code of [ "ECONNRESET", + "ECONNABORTED", "ETIMEDOUT", "EPIPE", "ENOTFOUND", @@ -158,6 +161,8 @@ describe("isRetryable: transport failures", () => { "ECONNREFUSED", "EHOSTUNREACH", "ENETUNREACH", + "ENETRESET", + "ENETDOWN", ]) { expect(isRetryable(errnoError(code))).toBe(true); } @@ -215,6 +220,27 @@ describe("isRetryable: transport failures", () => { looped.cause = looped; expect(isRetryable(looped)).toBe(false); }); + + it("terminates on a cause chain that loops through two errors", () => { + const first: Error & { cause?: unknown } = new Error("first"); + const second = new Error("second", { cause: first }); + first.cause = second; + expect(isRetryable(first)).toBe(false); + }); + + it("reads the error and at most seven causes below it", () => { + // The walk is bounded at eight links. An errno at the eighth link is + // found; one at the ninth is not. + const buried = (causes: number): Error => { + let err = errnoError("ECONNRESET"); + for (let i = 0; i < causes; i++) { + err = new Error(`wrapper ${i}`, { cause: err }); + } + return err; + }; + expect(isRetryable(buried(7))).toBe(true); + expect(isRetryable(buried(8))).toBe(false); + }); }); describe("isRetryable: stream truncation versus corruption", () => { @@ -279,10 +305,9 @@ describe("isSafeToReplay", () => { * that is not the whole question: the other half is "could the first * attempt already have taken effect on the server?". * - * quak's non-idempotent calls are `/users/srp/create-session`, - * `/users/two-factor/verify` (which consumes one of a limited number of - * 2FA attempts) and `/files/thumbnail`. A blind replay of any of them can - * do real damage, so they retry only on the failures that establish no TCP + * The calls this guards are the `POST` and `PUT` requests listed in the + * README under "Endpoints used". A blind replay of some of them can do + * real damage, so they retry only on the failures that establish no TCP * connection to the server ever existed — DNS produced no address, or the * peer refused the connection — and therefore that no request byte can * have been transmitted. @@ -333,6 +358,52 @@ describe("isSafeToReplay", () => { ).toBe(false); expect(isSafeToReplay(new TypeError("fetch failed"))).toBe(false); }); + + it("does not replay any other errno the classifier names", () => { + for (const code of [ + "ECONNRESET", + "ECONNABORTED", + "ETIMEDOUT", + "EPIPE", + "EHOSTUNREACH", + "ENETUNREACH", + "ENETRESET", + "ENETDOWN", + ]) { + expect(isSafeToReplay(errnoError(code))).toBe(false); + } + }); + + it("does not replay a chain that also shows the request may have gone out", () => { + // A connect errno somewhere in the chain is not enough: any other + // errno beside it is doubt, and doubt is not replayed. + const reset = Object.assign( + new Error("read ECONNRESET", { cause: errnoError("ECONNREFUSED") }), + { code: "ECONNRESET" }, + ); + const mixed = new TypeError("fetch failed", { cause: reset }); + expect(isRetryable(mixed)).toBe(true); + expect(isSafeToReplay(mixed)).toBe(false); + }); + + it("does not replay a chain longer than the walk reads", () => { + // Eight connect errnos, then a reset at the ninth link, below the + // limit. The walk never sees the reset, so it cannot rule it out. + const refusedChain = (below: Error | undefined): Error => { + let err = below; + for (let i = 0; i < 8; i++) { + err = Object.assign(new Error(`refused ${i}`, { cause: err }), { + code: "ECONNREFUSED", + }); + } + return err as Error; + }; + expect(isSafeToReplay(refusedChain(errnoError("ECONNRESET")))).toBe( + false, + ); + // The same eight links with nothing below them are replayable. + expect(isSafeToReplay(refusedChain(undefined))).toBe(true); + }); }); // --------------------------------------------------------------------------- diff --git a/test/smoke.test.ts b/test/smoke.test.ts index 77fe023..a33c747 100644 --- a/test/smoke.test.ts +++ b/test/smoke.test.ts @@ -1,9 +1,43 @@ -import { describe, expect, it } from "vitest"; +import { afterEach, describe, expect, it, vi } from "vitest"; +import { readFileSync } from "node:fs"; import { VERSION } from "../src/index.js"; -describe("quak", () => { - it("exports a version string", () => { - expect(typeof VERSION).toBe("string"); - expect(VERSION.length).toBeGreaterThan(0); +const packageVersion = ( + JSON.parse( + readFileSync(new URL("../package.json", import.meta.url), "utf-8"), + ) as { version: string } +).version; + +class ExitCalled extends Error {} + +describe("version", () => { + const argv = process.argv; + + afterEach(() => { + process.argv = argv; + vi.restoreAllMocks(); + }); + + it("exports the version from package.json", () => { + expect(VERSION).toBe(packageVersion); + }); + + // Runs bin/quak.ts with --version. commander prints the version and then + // calls process.exit, which is stubbed to throw so the test survives. + it("reports the version from package.json in quak --version", async () => { + const printed: string[] = []; + vi.spyOn(process.stdout, "write").mockImplementation((chunk) => { + printed.push(String(chunk)); + return true; + }); + vi.spyOn(process, "exit").mockImplementation(() => { + throw new ExitCalled(); + }); + process.argv = ["node", "quak", "--version"]; + + await expect(import("../bin/quak.js")).rejects.toBeInstanceOf( + ExitCalled, + ); + expect(printed.join("").trim()).toBe(packageVersion); }); }); diff --git a/test/thumbnails/thumbnails.test.ts b/test/thumbnails/thumbnails.test.ts index 940eaaa..580a312 100644 --- a/test/thumbnails/thumbnails.test.ts +++ b/test/thumbnails/thumbnails.test.ts @@ -215,6 +215,9 @@ const buildThumbMock = async (opts?: { thumbnail: { decryptionHeader: toBase64(sodium.randombytes_buf(24)), }, + // The encrypted size of the thumbnail the server records; large + // enough here that the default encoding fits. + info: { thumbSize: 1_000_000 }, updationTime: TEST_TIME, }; }; @@ -446,6 +449,32 @@ const openLib = (client: Client): Promise<Library> => precacheOriginals: false, }); +/** The mock's raw record for one file, for a test to change before login. */ +const rawFile = (m: ThumbMockState, fileID: number): Record<string, unknown> => + m.filesByCollection[1]!.find((f) => f.id === fileID)!; + +/** Replace the original the mock serves for one file. */ +const replaceOriginal = ( + m: ThumbMockState, + fileID: number, + body: Uint8Array, +): void => { + const push = sodium.crypto_secretstream_xchacha20poly1305_init_push( + m.fileKeys[fileID]!, + ); + m.fileCiphertexts[fileID] = + sodium.crypto_secretstream_xchacha20poly1305_push( + push.state, + body, + null, + sodium.crypto_secretstream_xchacha20poly1305_TAG_FINAL, + ); + rawFile(m, fileID).file = { decryptionHeader: toBase64(push.header) }; +}; + +const isOriginalDownload = (url: string): boolean => + url.includes("files.ente.io") || url.includes("/files/download/"); + const login = (fetch: typeof globalThis.fetch, retry?: RetryOptions) => Client.login({ email: TEST_EMAIL, @@ -581,6 +610,33 @@ describe("listMissingThumbnails", () => { // Should still be 2, not 4 (each file checked only once) expect(missing.length).toBe(2); }); + + it("skips a file another account owns without fetching its thumbnail", async () => { + const otherMock = await buildThumbMock(); + rawFile(otherMock, 102).ownerID = 7; + const logs: string[] = []; + const counted = countingFetch( + buildThumbFetch(otherMock), + (url) => url.includes("thumbnails.ente.io") && url.includes("102"), + ); + const client = await login(counted.fetch); + const lib = await openLib(client); + + const missing = await listMissingThumbnails(lib, client, (msg) => + logs.push(msg), + ); + lib.close(); + + expect(missing.map((m) => m.fileID)).toEqual([101]); + expect(counted.matched()).toBe(0); + expect( + logs.some( + (l) => + l.includes("Skipping file-102.jpg") && + l.includes("another account"), + ), + ).toBe(true); + }); }); describe("fixMissingThumbnails", () => { @@ -688,6 +744,93 @@ describe("fixMissingThumbnails", () => { expect(fixMock.uploadedThumbnails.length).toBe(1); expect(fixMock.uploadedThumbnails[0]!.fileID).toBe(101); }); + + it("skips a file another account owns without downloading it", async () => { + // The server accepts a thumbnail only from the file's owner. + const fixMock = await buildThumbMock(); + rawFile(fixMock, 101).ownerID = 7; + const counted = countingFetch( + buildThumbFetch(fixMock), + isOriginalDownload, + ); + const client = await login(counted.fetch); + const lib = await openLib(client); + + const results = await fixMissingThumbnails(lib, client, [101]); + lib.close(); + + expect(results[0]!.status).toBe("skipped"); + expect(results[0]!.reason).toContain("another account"); + expect(counted.matched()).toBe(0); + expect(fixMock.uploadedThumbnails.length).toBe(0); + }); + + it("skips a file whose recorded thumbnail size is 0 without downloading it", async () => { + // The server refuses a thumbnail larger than the one it records, and + // no thumbnail is 0 bytes. + const fixMock = await buildThumbMock(); + rawFile(fixMock, 101).info = { thumbSize: 0 }; + const counted = countingFetch( + buildThumbFetch(fixMock), + isOriginalDownload, + ); + const client = await login(counted.fetch); + const lib = await openLib(client); + + const results = await fixMissingThumbnails(lib, client, [101]); + lib.close(); + + expect(results[0]!.status).toBe("skipped"); + expect(results[0]!.reason).toContain("recorded thumbnail size is 0"); + expect(counted.matched()).toBe(0); + expect(fixMock.uploadedThumbnails.length).toBe(0); + }); + + it("re-encodes smaller until the thumbnail fits the recorded size", async () => { + // A noisy 400x300 JPEG, which the default encoding (quality 50, not + // resized because it is under 720 px) cannot compress below the size + // recorded here: one byte less than that encoding's ciphertext. + const fixMock = await buildThumbMock(); + const w = 400; + const h = 300; + const noisy = new Uint8Array( + jpegJs.encode( + { + data: sodium.randombytes_buf(w * h * 4), + width: w, + height: h, + }, + 90, + ).data, + ); + replaceOriginal(fixMock, 101, noisy); + const decoded = jpegJs.decode(noisy, { + useTArray: true, + formatAsRGBA: true, + }); + const defaultSize = + jpegJs.encode(decoded, 50).data.length + + sodium.crypto_secretstream_xchacha20poly1305_ABYTES; + const recordedSize = defaultSize - 1; + rawFile(fixMock, 101).info = { thumbSize: recordedSize }; + + const client = await login(buildThumbFetch(fixMock)); + const lib = await openLib(client); + + const results = await fixMissingThumbnails(lib, client, [101]); + lib.close(); + + expect(results[0]!.status).toBe("fixed"); + const upload = fixMock.uploadedThumbnails[0]!; + expect(upload.ciphertext.length).toBeLessThanOrEqual(recordedSize); + const decrypted = decryptBlob( + upload.ciphertext, + fromBase64(upload.decryptionHeader), + fixMock.fileKeys[101]!, + ); + expect(decrypted[0]).toBe(0xff); + expect(decrypted[1]).toBe(0xd8); + }); }); describe("Client.getApiClient", () => { diff --git a/vitest.config.ts b/vitest.config.ts new file mode 100644 index 0000000..e39bf41 --- /dev/null +++ b/vitest.config.ts @@ -0,0 +1,10 @@ +import { configDefaults, defineConfig } from "vitest/config"; + +// vitest does not read .gitignore when looking for tests. A checkout nested +// under .claude/ has its own test/ tree, and without this exclude the suite +// runs once per nested checkout and still reports success. +export default defineConfig({ + test: { + exclude: [...configDefaults.exclude, ".claude/**"], + }, +}); diff --git a/yarn.lock b/yarn.lock index ab845b6..7759c2d 100644 --- a/yarn.lock +++ b/yarn.lock @@ -528,13 +528,6 @@ resolved "https://registry.yarnpkg.com/@types/json-schema/-/json-schema-7.0.15.tgz#596a1747233694d50f6ad8a7869fcb6f56cf5841" integrity sha512-5+fP8P8MFNC+AyZCDxrB2pkZFPGzqQWUzpSeuuVLvm8VMcorNYavBqoFcxK8bQz4Qsbn4oUEEem4wDLfcysGHA== -"@types/libsodium-wrappers-sumo@0.8.2": - version "0.8.2" - resolved "https://registry.yarnpkg.com/@types/libsodium-wrappers-sumo/-/libsodium-wrappers-sumo-0.8.2.tgz#488e8747fbb982fe901020b5afeaddfa63da6830" - integrity sha512-uFOBpg/r21hExVlh2ty8YpDfSR+Yy3Jn8XS4+SSjitbhTxdYq+pBz/49XRxyUFe8SzqujHf/Wu0/O4d+FUtNfQ== - dependencies: - libsodium-wrappers-sumo "*" - "@types/node@22.18.13": version "22.18.13" resolved "https://registry.yarnpkg.com/@types/node/-/node-22.18.13.tgz#a037c4f474b860be660e05dbe92a9ef945472e28" @@ -1076,6 +1069,11 @@ fastq@^1.6.0: dependencies: reusify "^1.0.4" +fflate@0.8.3: + version "0.8.3" + resolved "https://registry.yarnpkg.com/fflate/-/fflate-0.8.3.tgz#bc27d8eb30343d4d512abb03480202ce65d825fc" + integrity sha512-tbZNuJrLwGUp3zshBtdy4W+ORxZuIh8a5ilyIEQDC5rY1f3U20JMry0Ll3WBzU58EZKsEuJFXhb5gwv8CsPvgA== + file-entry-cache@^8.0.0: version "8.0.0" resolved "https://registry.yarnpkg.com/file-entry-cache/-/file-entry-cache-8.0.0.tgz#7787bddcf1131bffb92636c69457bbc0edd6d81f" @@ -1249,7 +1247,7 @@ libsodium-sumo@^0.8.0: resolved "https://registry.yarnpkg.com/libsodium-sumo/-/libsodium-sumo-0.8.4.tgz#6d4687781fa0ad398af14a7df872d5c27cf8cd31" integrity sha512-TMtHShQfVVsaxDygyapvUC3o7YsPgXa/hRWeIgzyFz6w5k/1hirGptCxp1U7XwW3rCskaTTYKgV10v86UiGgNw== -libsodium-wrappers-sumo@*, libsodium-wrappers-sumo@0.8.4: +libsodium-wrappers-sumo@0.8.4: version "0.8.4" resolved "https://registry.yarnpkg.com/libsodium-wrappers-sumo/-/libsodium-wrappers-sumo-0.8.4.tgz#6656a3e7e0551ecce08ddee4bfb501a092eac6fa" integrity sha512-ql7hcgulKZ3ekfa2DGAogcCKsWU0diA/0nArz1CFzh93WQdb46/Kj18ka/Hifq6uA3Ush34Pc6vU/6HXeRwUkg==