Harden the JPEG EXIF scan against malformed input (closes #11)
check / check (push) Successful in 29s
check / check (push) Successful in 29s
The segment scan behind `backup-metadata --exif` now checks every segment length against the bytes that remain and stops on lengths under 2, so a truncated or corrupt original can neither throw nor loop. A malformed or unparseable EXIF segment is recorded as `imageMetadata.exifError`, and a failure to read the original as `imageMetadataError` in the per-file JSON, instead of the field being silently left out. Tests use short hand-built byte arrays. Model: opus-5-5
This commit was merged in pull request #84.
This commit is contained in:
+89
-66
@@ -15,18 +15,35 @@ export interface MetadataBackupOptions {
|
||||
onProgress?: ProgressCallback;
|
||||
}
|
||||
|
||||
// Extract the raw EXIF APP1 segment from JPEG bytes. Returns the EXIF
|
||||
// data buffer (starting after the APP1 length field, at the "Exif\0\0"
|
||||
// header) or undefined if no APP1 marker is found.
|
||||
const extractExifFromJpeg = (buf: Uint8Array): Buffer | undefined => {
|
||||
if (buf[0] !== 0xff || buf[1] !== 0xd8) return undefined;
|
||||
// Find the raw EXIF APP1 segment in JPEG bytes. Returns `exif` (the segment
|
||||
// data, starting at the "Exif\0\0" header) when there is one, nothing when the
|
||||
// bytes are not a JPEG or carry no EXIF, and `error` when the segment layout is
|
||||
// malformed. Each segment length is checked against the bytes that remain and
|
||||
// each step moves forward by at least 4 bytes, so the scan ends on any input.
|
||||
export const extractExifFromJpeg = (
|
||||
buf: Uint8Array,
|
||||
): { exif?: Buffer; error?: string } => {
|
||||
if (buf[0] !== 0xff || buf[1] !== 0xd8) return {};
|
||||
let offset = 2;
|
||||
while (offset < buf.length - 1) {
|
||||
if (buf[offset] !== 0xff) return undefined;
|
||||
while (offset < buf.length) {
|
||||
if (offset + 2 > buf.length)
|
||||
return { error: `truncated segment marker at byte ${offset}` };
|
||||
if (buf[offset] !== 0xff)
|
||||
return { error: `no segment marker at byte ${offset}` };
|
||||
const marker = buf[offset + 1]!;
|
||||
if (marker === 0xda) break; // start of scan, no more markers
|
||||
if (offset + 3 >= buf.length) break;
|
||||
if (marker === 0xda) return {}; // start of scan, no more markers
|
||||
if (offset + 4 > buf.length)
|
||||
return { error: `truncated segment length at byte ${offset}` };
|
||||
const len = (buf[offset + 2]! << 8) | buf[offset + 3]!;
|
||||
// The length counts its own two bytes, so anything under 2 is invalid.
|
||||
if (len < 2)
|
||||
return {
|
||||
error: `segment length ${len} at byte ${offset} is too small`,
|
||||
};
|
||||
if (offset + 2 + len > buf.length)
|
||||
return {
|
||||
error: `segment length ${len} at byte ${offset} runs past the end of the file`,
|
||||
};
|
||||
if (marker === 0xe1) {
|
||||
// APP1 — check for "Exif\0\0" header
|
||||
if (
|
||||
@@ -35,65 +52,70 @@ const extractExifFromJpeg = (buf: Uint8Array): Buffer | undefined => {
|
||||
buf[offset + 6] === 0x69 &&
|
||||
buf[offset + 7] === 0x66
|
||||
) {
|
||||
return Buffer.from(
|
||||
buf.buffer,
|
||||
buf.byteOffset + offset + 4,
|
||||
len - 2,
|
||||
);
|
||||
return {
|
||||
exif: Buffer.from(
|
||||
buf.buffer,
|
||||
buf.byteOffset + offset + 4,
|
||||
len - 2,
|
||||
),
|
||||
};
|
||||
}
|
||||
}
|
||||
offset += 2 + len;
|
||||
}
|
||||
return undefined;
|
||||
return { error: "file ends before the image data" };
|
||||
};
|
||||
|
||||
const extractImageMetadata = (
|
||||
// Extract dimensions, EXIF and XMP from a file's bytes. When the EXIF segment
|
||||
// is malformed or cannot be parsed, the record carries the reason in
|
||||
// `exifError`.
|
||||
export const extractImageMetadata = (
|
||||
fileBytes: Uint8Array,
|
||||
): Record<string, unknown> | undefined => {
|
||||
const result: Record<string, unknown> = {};
|
||||
|
||||
// Try to get dimensions from JPEG decode
|
||||
try {
|
||||
const result: Record<string, unknown> = {};
|
||||
|
||||
// Try to get dimensions from JPEG decode
|
||||
try {
|
||||
const decoded = jpeg.decode(fileBytes, {
|
||||
useTArray: true,
|
||||
formatAsRGBA: false,
|
||||
});
|
||||
result.format = "jpeg";
|
||||
result.width = decoded.width;
|
||||
result.height = decoded.height;
|
||||
} catch {
|
||||
// Not a JPEG or corrupt; still try EXIF extraction
|
||||
}
|
||||
|
||||
const exifBuf = extractExifFromJpeg(fileBytes);
|
||||
if (exifBuf) {
|
||||
try {
|
||||
result.exif = exifReader(exifBuf);
|
||||
} catch {
|
||||
result.exifRaw = exifBuf.toString("base64");
|
||||
}
|
||||
}
|
||||
|
||||
// Extract XMP (look for "http://ns.adobe.com/xap" in the bytes)
|
||||
const xmpStart = Buffer.from(fileBytes).indexOf("<?xpacket begin");
|
||||
if (xmpStart !== -1) {
|
||||
const xmpEnd = Buffer.from(fileBytes).indexOf(
|
||||
"<?xpacket end",
|
||||
xmpStart,
|
||||
);
|
||||
if (xmpEnd !== -1) {
|
||||
const end = Buffer.from(fileBytes).indexOf("?>", xmpEnd);
|
||||
result.xmp = Buffer.from(fileBytes)
|
||||
.subarray(xmpStart, end !== -1 ? end + 2 : xmpEnd + 50)
|
||||
.toString("utf-8");
|
||||
}
|
||||
}
|
||||
|
||||
return Object.keys(result).length > 0 ? result : undefined;
|
||||
const decoded = jpeg.decode(fileBytes, {
|
||||
useTArray: true,
|
||||
formatAsRGBA: false,
|
||||
});
|
||||
result.format = "jpeg";
|
||||
result.width = decoded.width;
|
||||
result.height = decoded.height;
|
||||
} catch {
|
||||
return undefined;
|
||||
// Not every original is a JPEG (PNG, HEIC, video), so a failed decode
|
||||
// is expected and only means no dimensions; a malformed JPEG is still
|
||||
// reported below through `exifError`.
|
||||
}
|
||||
|
||||
const { exif, error } = extractExifFromJpeg(fileBytes);
|
||||
if (error) result.exifError = error;
|
||||
if (exif) {
|
||||
try {
|
||||
result.exif = exifReader(exif);
|
||||
} catch (err) {
|
||||
result.exifRaw = exif.toString("base64");
|
||||
result.exifError = err instanceof Error ? err.message : String(err);
|
||||
}
|
||||
}
|
||||
|
||||
// Extract XMP (look for "http://ns.adobe.com/xap" in the bytes)
|
||||
const xmpStart = Buffer.from(fileBytes).indexOf("<?xpacket begin");
|
||||
if (xmpStart !== -1) {
|
||||
const xmpEnd = Buffer.from(fileBytes).indexOf(
|
||||
"<?xpacket end",
|
||||
xmpStart,
|
||||
);
|
||||
if (xmpEnd !== -1) {
|
||||
const end = Buffer.from(fileBytes).indexOf("?>", xmpEnd);
|
||||
result.xmp = Buffer.from(fileBytes)
|
||||
.subarray(xmpStart, end !== -1 ? end + 2 : xmpEnd + 50)
|
||||
.toString("utf-8");
|
||||
}
|
||||
}
|
||||
|
||||
return Object.keys(result).length > 0 ? result : undefined;
|
||||
};
|
||||
|
||||
// Read a file's original bytes through the library's content cache and extract
|
||||
@@ -103,13 +125,9 @@ const extractImageMetadata = (
|
||||
const extractExif = async (
|
||||
photo: Photo,
|
||||
): Promise<Record<string, unknown> | undefined> => {
|
||||
try {
|
||||
const { path } = await photo.original();
|
||||
const fileBytes = new Uint8Array(readFileSync(path));
|
||||
return extractImageMetadata(fileBytes);
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
const { path } = await photo.original();
|
||||
const fileBytes = new Uint8Array(readFileSync(path));
|
||||
return extractImageMetadata(fileBytes);
|
||||
};
|
||||
|
||||
// Dump every decrypted metadata layer the account holds into a directory tree
|
||||
@@ -215,8 +233,13 @@ export const runMetadataBackup = async (
|
||||
|
||||
if (wantExif && !writtenFileIDs.has(file.id)) {
|
||||
log(`[${file.metadata.title}] Extracting EXIF...`);
|
||||
const exifData = await extractExif(photo);
|
||||
if (exifData) fileMeta.imageMetadata = exifData;
|
||||
try {
|
||||
const exifData = await extractExif(photo);
|
||||
if (exifData) fileMeta.imageMetadata = exifData;
|
||||
} catch (err) {
|
||||
fileMeta.imageMetadataError =
|
||||
err instanceof Error ? err.message : String(err);
|
||||
}
|
||||
}
|
||||
writtenFileIDs.add(file.id);
|
||||
|
||||
|
||||
Reference in New Issue
Block a user