diff --git a/README.md b/README.md index b183973..dd10432 100644 --- a/README.md +++ b/README.md @@ -681,6 +681,11 @@ Under `cacheDirectory`: fetched.json per-file fetch bookkeeping ``` +When `metadata.json` belongs to a different account than the client's, +`Library.open` deletes it and `mldata/` and starts from an empty cache. Cached +originals and thumbnails are kept; they are reached only through the files the +current account's records name. + A stored file appears only via an atomic temp-then-rename, so its presence means it is complete. The design also calls for a content-hash comparison against `FileMetadata.hash` on each fetched original; that check is deferred (issue diff --git a/TODO.md b/TODO.md index 5bfdace..dcc9eb8 100644 --- a/TODO.md +++ b/TODO.md @@ -18,12 +18,17 @@ Tag v1.0.0. # Completed Steps +- 2026-09-23: Kept one account's cache from mixing with another's (issue 104). + When `metadata.json` in the cache directory was written for a different, + non-zero user ID than the client's, `Library.open` deletes it and `mldata/` + and starts empty, so the first refresh enumerates from 0. This only happens + with `--cache-dir` or an explicit `cacheDirectory`; the default path already + includes the user ID. A test opens one account's cache as another account. - 2026-09-23: `backup-metadata` no longer stops on one failed ML data request (issue 101). Each request of up to 200 files is tried on its own; a failed one is logged, its files are written with the reason in `mlDataError`, and the command exits 1 once the dump is complete. `fetchMLData`, which only this command used, is gone; the command calls `fetchMLDataBatch` per batch. - - 2026-09-23: Single-sourced the version string (issue 5). `package.json` is the only place it is written: `src/index.ts` imports it for `VERSION` and `bin/quak.ts` passes `VERSION` to commander. tsc copies `package.json` to diff --git a/src/library/index.ts b/src/library/index.ts index df1199c..f61bd47 100644 --- a/src/library/index.ts +++ b/src/library/index.ts @@ -26,6 +26,7 @@ // store marked unsaved until a later save actually lands, so a stuck disk is // never masked by a subsequent empty refresh. +import { rm } from "node:fs/promises"; import { join } from "node:path"; import envPaths from "env-paths"; @@ -345,9 +346,20 @@ export class Library { const cacheDirectory = opts.cacheDirectory ?? join(envPaths("quak", { suffix: "" }).cache, String(userID)); - const store = await MetadataStore.load( - join(cacheDirectory, "metadata.json"), - ); + const metadataPath = join(cacheDirectory, "metadata.json"); + let store = await MetadataStore.load(metadataPath); + // A cache directory given explicitly can hold another account's cache. + // Its records and cursor are not this account's, so delete it and the + // ML data beside it and start empty. A user ID of 0 means the cache + // was never refreshed and so holds nothing to discard. + if (store.userID !== 0 && store.userID !== userID) { + await rm(metadataPath, { force: true }); + await rm(join(cacheDirectory, "mldata"), { + recursive: true, + force: true, + }); + store = await MetadataStore.load(metadataPath); + } const intervalMs = (opts.refreshIntervalSeconds ?? DEFAULT_REFRESH_INTERVAL_SECONDS) * 1000; diff --git a/test/library/mldata.test.ts b/test/library/mldata.test.ts index 1dfa0c4..3de1cdc 100644 --- a/test/library/mldata.test.ts +++ b/test/library/mldata.test.ts @@ -13,6 +13,8 @@ * 2. `Library` wiring: after each refresh the library fetches ML data through * the metadata pool for every known file not yet cached, is incremental on * later refreshes, and refetches a file whose `updationTime` advanced. + * Opening a cache directory written by another account starts empty, + * its ML data included (issue #104). * * Embedding values are chosen to be exactly representable as float32 so the * round-trip through `clip.f32` compares equal. @@ -498,4 +500,63 @@ describe("Library ML-data fetch on refresh", () => { await lib.close(); } }); + + it("starts empty when the cache directory holds another account's cache", async () => { + // Account A fills the cache directory: metadata and ML data. + const clientA = new MLMockClient(); + clientA.collectionsQueue.push({ + collections: [collection(1, 100)], + deleted: [], + cursor: 100, + }); + clientA.filesFor(1, { + files: [file(1001, 1, 90)], + deleted: [], + cursor: 90, + }); + clientA.mlByFile.set(1001, payload([0.5, 0.25, 0.75])); + const libA = await Library.open({ + client: clientA, + cacheDirectory, + refreshIntervalSeconds: 3600, + }); + try { + await vi.waitFor( + () => expect(libA.status().lastMLFetchAt).toBeGreaterThan(0), + { timeout: 2000, interval: 5 }, + ); + } finally { + await libA.close(); + } + + // Account B opens the same directory. + const clientB = new MLMockClient(); + clientB.userID = USER_ID + 1; + const sinceTimes: number[] = []; + const realCollectionsSince = clientB.collectionsSince.bind(clientB); + clientB.collectionsSince = async (args) => { + sinceTimes.push(args.sinceTime); + return realCollectionsSince(args); + }; + const libB = await Library.open({ + client: clientB, + cacheDirectory, + refreshIntervalSeconds: 3600, + }); + try { + expect(sinceTimes[0]).toBe(0); + expect(libB.status().userID).toBe(USER_ID + 1); + expect(libB.listCollections()).toEqual([]); + expect(libB.getFile(1, 1001)).toBeUndefined(); + expect(await libB.mldata.forFile({ fileID: 1001 })).toBeUndefined(); + expect( + libB.mldata.searchByEmbedding({ embedding: [0.5, 0.25, 0.75] }), + ).toEqual([]); + expect( + existsSync(join(cacheDirectory, "mldata", "1001.json")), + ).toBe(false); + } finally { + await libB.close(); + } + }); });