Files
quak/src/metadata-backup.ts
T
clawbot 67d554fb46
check / check (push) Successful in 2m10s
exif(): read HEIF/HEIC originals with exifreader (closes #145)
`photo.exif()` and `quak backup-metadata --exif` now read EXIF through `exifreader`. HEIC/HEIF originals get EXIF, including a live photo's image, as do the other formats `exifreader` reads. It replaces `exif-reader` and the hand-written JPEG scan. `PhotoExif` is unchanged.

The `backup-metadata` dump now holds `exifreader`'s tag output, with unnamed tags keyed `undefined-` plus their number. GPS altitude without a reference counts as above sea level. A latitude or longitude without its hemisphere tag, an unreadable text tag, and a date the parser rejects each give no field.

Licence: `exifreader` is MPL-2.0, used unmodified.

Model: opus-5-5
2026-10-01 22:34:09 +02:00

234 lines
8.7 KiB
TypeScript

import { mkdirSync, readFileSync, writeFileSync } from "node:fs";
import { join } from "node:path";
import * as jpeg from "jpeg-js";
import type { Client } from "./client.js";
import { readExifTags } from "./exif.js";
import type { Library, Photo } from "./library/index.js";
import { sanitizeFileName } from "./filename.js";
import {
fetchMLDataBatch,
MLDATA_BATCH_SIZE,
type MLData,
} from "./mldata-fetch.js";
import type { EnteFile } from "./model/types.js";
export type ProgressCallback = (message: string) => void;
export interface MetadataBackupOptions {
exif?: boolean;
onProgress?: ProgressCallback;
}
// Extract dimensions, EXIF and XMP from a file's bytes. `exif` is the EXIF tags
// exifreader returns, from any image format it reads. When it finds an EXIF
// block but reads no tag from it, the record keeps the block's bytes, base64,
// in `exifRaw`, with the reason in `exifError`.
export const extractImageMetadata = (
fileBytes: Uint8Array,
): Record<string, unknown> | undefined => {
const result: Record<string, unknown> = {};
// Try to get dimensions from JPEG decode
try {
const decoded = jpeg.decode(fileBytes, {
useTArray: true,
formatAsRGBA: false,
});
result.format = "jpeg";
result.width = decoded.width;
result.height = decoded.height;
} catch {
// Not every original is a JPEG (PNG, HEIC, video), so a failed decode
// is expected and only means no dimensions; unreadable EXIF is still
// reported below through `exifError`.
}
const tags = readExifTags(fileBytes);
if (tags?.exif && Object.keys(tags.exif).length > 0) {
result.exif = tags.exif;
} else if (tags?.exif) {
const block = tags.metadataRange?.blocks.find((b) => b.type === "exif");
if (block) {
result.exifRaw = Buffer.from(
fileBytes.subarray(block.start, block.end),
).toString("base64");
}
result.exifError = "no tag could be read from the EXIF block";
}
// Extract XMP (look for "http://ns.adobe.com/xap" in the bytes)
const xmpStart = Buffer.from(fileBytes).indexOf("<?xpacket begin");
if (xmpStart !== -1) {
const xmpEnd = Buffer.from(fileBytes).indexOf(
"<?xpacket end",
xmpStart,
);
if (xmpEnd !== -1) {
const end = Buffer.from(fileBytes).indexOf("?>", xmpEnd);
result.xmp = Buffer.from(fileBytes)
.subarray(xmpStart, end !== -1 ? end + 2 : xmpEnd + 50)
.toString("utf-8");
}
}
return Object.keys(result).length > 0 ? result : undefined;
};
// Read a file's original bytes through the library's content cache and extract
// its embedded image metadata. The bytes come from `photo.original()` — the
// same on-disk cache the rest of the library fills — rather than a fresh
// per-call download to a throwaway temp file. For a live photo, its `path` is
// the image.
const extractExif = async (
photo: Photo,
): Promise<Record<string, unknown> | undefined> => {
const { path } = await photo.original();
const fileBytes = new Uint8Array(readFileSync(path));
return extractImageMetadata(fileBytes);
};
// Dump every decrypted metadata layer the account holds into a directory tree
// of plain JSON: account, per-collection, and per-file records including the
// private and public magic metadata and (by default) the ML data. Collections
// and files are enumerated from the library's cache, which the caller refreshes
// first. Returns how many ML data requests failed; their files are still
// written, with `mlDataError` in place of `mlData`.
export const runMetadataBackup = async (
lib: Library,
client: Client,
outDir: string,
opts?: MetadataBackupOptions,
): Promise<{ failedMLBatches: number }> => {
const log = opts?.onProgress ?? (() => {});
const wantExif = opts?.exif ?? false;
mkdirSync(outDir, { recursive: true });
mkdirSync(join(outDir, "collections"), { recursive: true });
const { email, userID } = client.whoami();
writeFileSync(
join(outDir, "account.json"),
JSON.stringify({ email, userID }, null, 2),
);
log("Fetching collections...");
// Enumerate through the library's read surface. Each album carries its
// photos, but the full decrypted `Collection`/`EnteFile` records (with the
// magic-metadata layers this dump exists to preserve) come from the
// library's by-id accessors.
const allFiles: { file: EnteFile; photo: Photo; colDirName: string }[] = [];
const fileKeys = new Map<number, Uint8Array>();
const seenFileIDs = new Set<number>();
for (const album of lib.albums.list()) {
const col = lib.getCollection(album.collectionID);
if (!col) continue;
const dirName = `${col.id}-${sanitizeFileName(col.name, "unnamed")}`;
const colDir = join(outDir, "collections", dirName);
mkdirSync(colDir, { recursive: true });
const collectionMeta: Record<string, unknown> = {
id: col.id,
name: col.name,
type: col.type,
ownerID: col.ownerID,
isShared: col.isShared,
updationTime: col.updationTime,
};
if (col.magicMetadata) collectionMeta.magicMetadata = col.magicMetadata;
if (col.pubMagicMetadata)
collectionMeta.pubMagicMetadata = col.pubMagicMetadata;
if (col.sharedMagicMetadata)
collectionMeta.sharedMagicMetadata = col.sharedMagicMetadata;
writeFileSync(
join(colDir, "_collection.json"),
JSON.stringify(collectionMeta, null, 2),
);
log(`[${col.name}] Fetching files...`);
const photos = album.photos.list();
log(`[${col.name}] ${photos.length} file(s)`);
for (const photo of photos) {
const file = lib.getFile(col.id, photo.fileID);
if (!file) continue;
allFiles.push({ file, photo, colDirName: dirName });
if (!seenFileIDs.has(file.id)) {
fileKeys.set(file.id, file.key);
seenFileIDs.add(file.id);
}
}
}
// One failed request (retries exhausted) must not end the dump: its files
// get the reason in `mlDataError` and the other batches go on.
log("Fetching ML data (face detections, CLIP embeddings)...");
const mlDataMap = new Map<number, MLData>();
const mlDataErrors = new Map<number, string>();
let failedMLBatches = 0;
const fileIDs = [...fileKeys.keys()];
for (let i = 0; i < fileIDs.length; i += MLDATA_BATCH_SIZE) {
const batch = fileIDs.slice(i, i + MLDATA_BATCH_SIZE);
try {
const result = await fetchMLDataBatch(
client.getApiClient(),
batch,
fileKeys,
);
for (const [id, payload] of result) mlDataMap.set(id, payload);
} catch (err) {
const reason = err instanceof Error ? err.message : String(err);
failedMLBatches++;
log(
`ML data request for ${batch.length} file(s) failed: ${reason}`,
);
for (const id of batch) mlDataErrors.set(id, reason);
}
}
log(`Got ML data for ${mlDataMap.size} file(s)`);
const writtenFileIDs = new Set<number>();
for (const { file, photo, colDirName } of allFiles) {
const colDir = join(outDir, "collections", colDirName);
const fileMeta: Record<string, unknown> = {
id: file.id,
collectionID: file.collectionID,
ownerID: file.ownerID,
metadata: file.metadata,
updationTime: file.updationTime,
};
if (file.magicMetadata) fileMeta.magicMetadata = file.magicMetadata;
if (file.pubMagicMetadata)
fileMeta.pubMagicMetadata = file.pubMagicMetadata;
const ml = mlDataMap.get(file.id);
if (ml) fileMeta.mlData = ml;
const mlError = mlDataErrors.get(file.id);
if (mlError) fileMeta.mlDataError = mlError;
if (wantExif && !writtenFileIDs.has(file.id)) {
log(`[${file.metadata.title}] Extracting EXIF...`);
try {
const exifData = await extractExif(photo);
if (exifData) fileMeta.imageMetadata = exifData;
} catch (err) {
fileMeta.imageMetadataError =
err instanceof Error ? err.message : String(err);
}
}
writtenFileIDs.add(file.id);
writeFileSync(
join(colDir, `${file.id}.json`),
JSON.stringify(fileMeta, null, 2),
);
}
log("Metadata backup complete.");
return { failedMLBatches };
};