import { mkdirSync, readFileSync, writeFileSync } from "node:fs"; import { join } from "node:path"; import * as jpeg from "jpeg-js"; import type { Client } from "./client.js"; import { readExifTags } from "./exif.js"; import type { Library, Photo } from "./library/index.js"; import { sanitizeFileName } from "./filename.js"; import { fetchMLDataBatch, MLDATA_BATCH_SIZE, type MLData, } from "./mldata-fetch.js"; import type { EnteFile } from "./model/types.js"; export type ProgressCallback = (message: string) => void; export interface MetadataBackupOptions { exif?: boolean; onProgress?: ProgressCallback; } // Extract dimensions, EXIF and XMP from a file's bytes. `exif` is the EXIF tags // exifreader returns, from any image format it reads. When it finds an EXIF // block but reads no tag from it, the record keeps the block's bytes, base64, // in `exifRaw`, with the reason in `exifError`. export const extractImageMetadata = ( fileBytes: Uint8Array, ): Record | undefined => { const result: Record = {}; // Try to get dimensions from JPEG decode try { const decoded = jpeg.decode(fileBytes, { useTArray: true, formatAsRGBA: false, }); result.format = "jpeg"; result.width = decoded.width; result.height = decoded.height; } catch { // Not every original is a JPEG (PNG, HEIC, video), so a failed decode // is expected and only means no dimensions; unreadable EXIF is still // reported below through `exifError`. } const tags = readExifTags(fileBytes); if (tags?.exif && Object.keys(tags.exif).length > 0) { result.exif = tags.exif; } else if (tags?.exif) { const block = tags.metadataRange?.blocks.find((b) => b.type === "exif"); if (block) { result.exifRaw = Buffer.from( fileBytes.subarray(block.start, block.end), ).toString("base64"); } result.exifError = "no tag could be read from the EXIF block"; } // Extract XMP (look for "http://ns.adobe.com/xap" in the bytes) const xmpStart = Buffer.from(fileBytes).indexOf("", xmpEnd); result.xmp = Buffer.from(fileBytes) .subarray(xmpStart, end !== -1 ? end + 2 : xmpEnd + 50) .toString("utf-8"); } } return Object.keys(result).length > 0 ? result : undefined; }; // Read a file's original bytes through the library's content cache and extract // its embedded image metadata. The bytes come from `photo.original()` — the // same on-disk cache the rest of the library fills — rather than a fresh // per-call download to a throwaway temp file. For a live photo, its `path` is // the image. const extractExif = async ( photo: Photo, ): Promise | undefined> => { const { path } = await photo.original(); const fileBytes = new Uint8Array(readFileSync(path)); return extractImageMetadata(fileBytes); }; // Dump every decrypted metadata layer the account holds into a directory tree // of plain JSON: account, per-collection, and per-file records including the // private and public magic metadata and (by default) the ML data. Collections // and files are enumerated from the library's cache, which the caller refreshes // first. Returns how many ML data requests failed; their files are still // written, with `mlDataError` in place of `mlData`. export const runMetadataBackup = async ( lib: Library, client: Client, outDir: string, opts?: MetadataBackupOptions, ): Promise<{ failedMLBatches: number }> => { const log = opts?.onProgress ?? (() => {}); const wantExif = opts?.exif ?? false; mkdirSync(outDir, { recursive: true }); mkdirSync(join(outDir, "collections"), { recursive: true }); const { email, userID } = client.whoami(); writeFileSync( join(outDir, "account.json"), JSON.stringify({ email, userID }, null, 2), ); log("Fetching collections..."); // Enumerate through the library's read surface. Each album carries its // photos, but the full decrypted `Collection`/`EnteFile` records (with the // magic-metadata layers this dump exists to preserve) come from the // library's by-id accessors. const allFiles: { file: EnteFile; photo: Photo; colDirName: string }[] = []; const fileKeys = new Map(); const seenFileIDs = new Set(); for (const album of lib.albums.list()) { const col = lib.getCollection(album.collectionID); if (!col) continue; const dirName = `${col.id}-${sanitizeFileName(col.name, "unnamed")}`; const colDir = join(outDir, "collections", dirName); mkdirSync(colDir, { recursive: true }); const collectionMeta: Record = { id: col.id, name: col.name, type: col.type, ownerID: col.ownerID, isShared: col.isShared, updationTime: col.updationTime, }; if (col.magicMetadata) collectionMeta.magicMetadata = col.magicMetadata; if (col.pubMagicMetadata) collectionMeta.pubMagicMetadata = col.pubMagicMetadata; if (col.sharedMagicMetadata) collectionMeta.sharedMagicMetadata = col.sharedMagicMetadata; writeFileSync( join(colDir, "_collection.json"), JSON.stringify(collectionMeta, null, 2), ); log(`[${col.name}] Fetching files...`); const photos = album.photos.list(); log(`[${col.name}] ${photos.length} file(s)`); for (const photo of photos) { const file = lib.getFile(col.id, photo.fileID); if (!file) continue; allFiles.push({ file, photo, colDirName: dirName }); if (!seenFileIDs.has(file.id)) { fileKeys.set(file.id, file.key); seenFileIDs.add(file.id); } } } // One failed request (retries exhausted) must not end the dump: its files // get the reason in `mlDataError` and the other batches go on. log("Fetching ML data (face detections, CLIP embeddings)..."); const mlDataMap = new Map(); const mlDataErrors = new Map(); let failedMLBatches = 0; const fileIDs = [...fileKeys.keys()]; for (let i = 0; i < fileIDs.length; i += MLDATA_BATCH_SIZE) { const batch = fileIDs.slice(i, i + MLDATA_BATCH_SIZE); try { const result = await fetchMLDataBatch( client.getApiClient(), batch, fileKeys, ); for (const [id, payload] of result) mlDataMap.set(id, payload); } catch (err) { const reason = err instanceof Error ? err.message : String(err); failedMLBatches++; log( `ML data request for ${batch.length} file(s) failed: ${reason}`, ); for (const id of batch) mlDataErrors.set(id, reason); } } log(`Got ML data for ${mlDataMap.size} file(s)`); const writtenFileIDs = new Set(); for (const { file, photo, colDirName } of allFiles) { const colDir = join(outDir, "collections", colDirName); const fileMeta: Record = { id: file.id, collectionID: file.collectionID, ownerID: file.ownerID, metadata: file.metadata, updationTime: file.updationTime, }; if (file.magicMetadata) fileMeta.magicMetadata = file.magicMetadata; if (file.pubMagicMetadata) fileMeta.pubMagicMetadata = file.pubMagicMetadata; const ml = mlDataMap.get(file.id); if (ml) fileMeta.mlData = ml; const mlError = mlDataErrors.get(file.id); if (mlError) fileMeta.mlDataError = mlError; if (wantExif && !writtenFileIDs.has(file.id)) { log(`[${file.metadata.title}] Extracting EXIF...`); try { const exifData = await extractExif(photo); if (exifData) fileMeta.imageMetadata = exifData; } catch (err) { fileMeta.imageMetadataError = err instanceof Error ? err.message : String(err); } } writtenFileIDs.add(file.id); writeFileSync( join(colDir, `${file.id}.json`), JSON.stringify(fileMeta, null, 2), ); } log("Metadata backup complete."); return { failedMLBatches }; };