Revert ML search (#50) to restore a green next (build fix pending)
check / check (push) Successful in 18s

Reverts the #50 merge (224bd101): mlsearch.ts failed tsc in the CI build (make check does not run make build, so it slipped past review). next restored to green; #50 to be redone with make build in its gate.

Model: opus-4-8
This commit was merged in pull request #69.
This commit is contained in:
2026-09-22 18:21:54 +02:00
parent 224bd101ab
commit 61dfec8d38
3 changed files with 4 additions and 240 deletions
+4 -12
View File
@@ -41,7 +41,6 @@ import {
type PhotosAPI,
type TimelineAPI,
} from "./read.js";
import { makeMLDataAPI, type MLDataAPI } from "./mlsearch.js";
export {
Album,
@@ -53,7 +52,6 @@ export {
type TimelineGroup,
type GroupBy,
} from "./read.js";
export { type MLDataAPI, type SimilarResult } from "./mlsearch.js";
import type { CollectionsPage, FilesPage } from "../client.js";
import { MLDATA_BATCH_SIZE, type MLData } from "../mldata-fetch.js";
import type { Collection, EnteFile } from "../model/types.js";
@@ -140,10 +138,6 @@ export class Library {
readonly albums: AlbumsAPI;
readonly photos: PhotosAPI;
readonly timeline: TimelineAPI;
// The content-similarity search surface over the CLIP index (issue #50).
// Present whether or not ML fetching is enabled; with no ML store it
// returns empty results.
readonly mldata: MLDataAPI;
private readonly client: LibraryClient;
private readonly store: MetadataStore;
@@ -152,7 +146,7 @@ export class Library {
private readonly onProgress?: RefreshProgressCallback;
private readonly pools: RequestPools;
// The ML-data cache, present only when the client can fetch ML data.
private readonly mlStore?: MLDataStore;
private readonly mldata?: MLDataStore;
private timer?: ReturnType<typeof setTimeout>;
private refreshing = false;
@@ -194,7 +188,7 @@ export class Library {
this.intervalMs = args.intervalMs;
this.onProgress = args.onProgress;
this.pools = args.pools;
this.mlStore = args.mldata;
this.mldata = args.mldata;
this.lastRecords = this.deriveNow();
// The read namespaces derive fresh from the store on each call, so they
@@ -203,8 +197,6 @@ export class Library {
this.albums = makeAlbumsAPI(derive);
this.photos = makePhotosAPI(derive);
this.timeline = makeTimelineAPI(derive);
// Reads the ML store live so results grow as ML data is fetched.
this.mldata = makeMLDataAPI(() => this.mlStore);
}
// Load the cache and start the refresh loop. With an empty cache the first
@@ -301,7 +293,7 @@ export class Library {
for (const c of collections) {
files += this.store.listFiles(c.id).length;
}
const ml = this.mlStore?.stats();
const ml = this.mldata?.stats();
return {
userID: this.store.userID,
collections: collections.length,
@@ -456,7 +448,7 @@ export class Library {
// advanced), through the metadata pool, and update the CLIP index. Guarded
// so passes never overlap; a failure is reported, not thrown.
private async runMLFetch(): Promise<void> {
const mldata = this.mlStore;
const mldata = this.mldata;
// Bind so the call keeps the client as its receiver when invoked
// through the pool below.
const fetchMLData = this.client.fetchMLData?.bind(this.client);
-120
View File
@@ -1,120 +0,0 @@
// The content-similarity search surface over the CLIP index (issue #50).
//
// This is `lib.mldata`. It answers three questions against the ML-data cache
// (#49) without touching the network:
//
// - `forFile` returns the whole stored payload (face boxes, landmarks,
// embedding) for a file, read from disk on demand — the only method here
// that touches the disk, and the only one that is async.
// - `similar` and `searchByEmbedding` rank fileIDs by cosine similarity over
// the packed `Float32Array` index alone. That index (~50k×512) already
// lives in RAM, so each query is a plain loop over it and nothing else.
//
// quak bundles no text encoder (owner-deferred), so `searchByEmbedding` takes
// the query vector the caller has produced elsewhere; `similar` uses the
// query file's own indexed embedding.
import type { MLData } from "../mldata-fetch.js";
import type { MLDataStore, MLIndex } from "./mldata.js";
// How many nearest files a query returns when the caller names no limit.
const DEFAULT_LIMIT = 20;
// One ranked result: a fileID and its cosine similarity to the query, in
// [-1, 1]. Callers wanting only the ids read `.fileID`.
export interface SimilarResult {
fileID: number;
score: number;
}
export interface MLDataAPI {
// The whole stored ML payload for a file, or undefined when it is not
// cached. Reads the payload from disk, so it is async.
forFile(args: { fileID: number }): Promise<MLData | undefined>;
// The files nearest the given file by cosine over their CLIP embeddings,
// most similar first, excluding the file itself. Empty when the file has
// no indexed embedding.
similar(args: { fileID: number; limit?: number }): SimilarResult[];
// The files nearest a caller-supplied query embedding by cosine, most
// similar first. Empty when the query is the wrong length for the index,
// has zero magnitude, or the index is empty.
searchByEmbedding(args: {
embedding: ArrayLike<number>;
limit?: number;
}): SimilarResult[];
}
// Rank the packed index by cosine similarity to `query`, most similar first,
// and return the top `limit`. `skip` (a query file's own id) is left out. Both
// each row's magnitude and the query's are computed here rather than cached:
// the index mutates as ML data is fetched, and one plain pass over ~50k×512
// floats is fast enough that a norm cache would only add a staleness bug. A
// zero-magnitude vector has no direction, so it is dropped rather than divided
// by zero.
const topByCosine = (
index: MLIndex,
query: ArrayLike<number>,
limit: number,
skip?: number,
): SimilarResult[] => {
const { fileIDs, embeddingLength, embeddings } = index;
if (embeddingLength === 0 || query.length !== embeddingLength) return [];
let queryNorm = 0;
for (let k = 0; k < embeddingLength; k++) queryNorm += query[k] * query[k];
queryNorm = Math.sqrt(queryNorm);
if (queryNorm === 0) return [];
const results: SimilarResult[] = [];
for (let i = 0; i < fileIDs.length; i++) {
const id = fileIDs[i];
if (id === skip) continue;
const base = i * embeddingLength;
let dot = 0;
let norm = 0;
for (let k = 0; k < embeddingLength; k++) {
const v = embeddings[base + k];
dot += query[k] * v;
norm += v * v;
}
if (norm === 0) continue;
results.push({
fileID: id,
score: dot / (queryNorm * Math.sqrt(norm)),
});
}
// Descending score, ties broken by ascending fileID for a stable order.
results.sort((a, b) => b.score - a.score || a.fileID - b.fileID);
return results.slice(0, Math.max(0, Math.trunc(limit)));
};
// Build the search surface over a store the library supplies lazily (the store
// is absent when the client cannot fetch ML data). Reading it per call keeps
// the surface current as the index grows.
export const makeMLDataAPI = (
store: () => MLDataStore | undefined,
): MLDataAPI => ({
forFile: ({ fileID }): Promise<MLData | undefined> => {
const s = store();
return s ? s.readPayload(fileID) : Promise.resolve(undefined);
},
similar: ({ fileID, limit }): SimilarResult[] => {
const s = store();
if (!s) return [];
const index = s.getIndex();
const pos = index.fileIDs.indexOf(fileID);
if (pos < 0) return [];
const base = pos * index.embeddingLength;
const query = index.embeddings.subarray(
base,
base + index.embeddingLength,
);
return topByCosine(index, query, limit ?? DEFAULT_LIMIT, fileID);
},
searchByEmbedding: ({ embedding, limit }): SimilarResult[] => {
const s = store();
if (!s) return [];
return topByCosine(s.getIndex(), embedding, limit ?? DEFAULT_LIMIT);
},
});