From 170252cfc915ac8448cce3e9aa2ff6c9c463e435 Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Sun, 9 Aug 2026 21:09:47 +0200 Subject: [PATCH] Locate embedded previews by parsing the container header MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Remote browsing must not transfer whole RAW files. The obvious shortcut — fetch a fixed prefix and hope the preview is inside — does not work: an embedded JPEG typically starts a few hundred KB in and runs for one to three MB, so a truncated fetch yields a JPEG whose scanlines stop partway down. Decoders render what they have rather than erroring, so the failure looks like a corrupt image rather than a short read. This reads the TIFF-structured containers — CR2, NEF, ARW, DNG, ORF — and returns an exact byte range for a Range: request. CR3 is ISO-BMFF and declines to the caller's whole-file path, which is correct if slower; a locator returning a wrong range would be far worse than one that declines. Every offset read from the file is bounds-checked against the real file length rather than trusted (NFR-SEC-1). Assisted-by: LLM --- core/dr-decode/src/locate.rs | 886 +++++++++++++++++++++++++++++++++++ 1 file changed, 886 insertions(+) create mode 100644 core/dr-decode/src/locate.rs diff --git a/core/dr-decode/src/locate.rs b/core/dr-decode/src/locate.rs new file mode 100644 index 0000000..537ea17 --- /dev/null +++ b/core/dr-decode/src/locate.rs @@ -0,0 +1,886 @@ +//! TRACES: FR-NC-3 | FR-CULL-2 +//! Finding an embedded preview's byte range from a file header alone. +//! +//! # Why this exists +//! +//! Remote browsing must not transfer whole RAW files (FR-NC-3). The obvious +//! shortcut — fetch a fixed prefix and hope the preview is inside it — does +//! not work: an embedded JPEG typically starts a few hundred KB in and runs +//! for one to three MB, so a truncated fetch yields a JPEG whose scanlines +//! stop partway down. Decoders render what they have rather than erroring, so +//! the failure looks like a corrupt image, not a short read. +//! +//! What FR-NC-3 actually specifies is two-stage: read the header, *parse the +//! container* to locate the preview, then fetch exactly those bytes. +//! +//! # Scope +//! +//! This reads TIFF-structured containers — CR2, NEF, ARW, DNG, and ORF all +//! carry their previews in IFD entries. CR3 is ISO-BMFF and is not handled +//! here; it falls back to the caller's whole-file path, which is correct if +//! slower. A locator that returned a wrong range would be far worse than one +//! that declines. +//! +//! Every offset read from the file is treated as hostile (NFR-SEC-1): bounds +//! are checked against the real file length, never trusted. + +use std::ops::Range; + +/// Where a preview lives inside its container. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct PreviewLocation { + /// Byte range of the JPEG, ready to hand to a `Range:` request. + pub range: Range, + /// Pixel dimensions where the container declared them. Used to pick the + /// largest preview that is still smaller than a full decode. + pub width: Option, + pub height: Option, +} + +impl PreviewLocation { + pub fn len(&self) -> u64 { + self.range.end - self.range.start + } + + pub fn is_empty(&self) -> bool { + self.range.end <= self.range.start + } +} + +/// How many bytes of header a caller should fetch before calling this. +/// +/// Large enough to cover the IFD chain in the formats measured, small enough +/// that a miss costs little. Metadata alone needs less, but IFD1/IFD2 entries +/// for the preview sit further in on some bodies. +pub const HEADER_BYTES: u64 = 256 * 1024; + +/// Locate the largest embedded preview at or below `max_edge`, if any. +/// +/// `header` is the first [`HEADER_BYTES`] of the file; `file_len` is the whole +/// file's length, needed to reject offsets that point past the end. +/// +/// Returns `None` when the container is not TIFF-structured, declares no +/// preview, or declares one whose range is not credible. +pub fn locate_preview(header: &[u8], file_len: u64) -> Option { + let tiff = TiffReader::new(header)?; + let mut best: Option = None; + + for ifd_offset in tiff.ifd_offsets() { + let Some(entries) = tiff.read_ifd(ifd_offset) else { + continue; + }; + + if let Some(loc) = preview_from_entries(&tiff, &entries, file_len) { + // Prefer the largest, since a bigger preview downscales better — + // but anything is better than nothing. + let better = match (&best, &loc) { + (None, _) => true, + (Some(b), l) => l.len() > b.len(), + }; + if better { + best = Some(loc); + } + } + } + + best +} + +/// TIFF tags that carry a preview's location. +mod tag { + /// Legacy thumbnail offset/length (IFD1 in most makes). + pub const JPEG_INTERCHANGE_FORMAT: u16 = 0x0201; + pub const JPEG_INTERCHANGE_FORMAT_LENGTH: u16 = 0x0202; + /// Strip-based storage, which is how DNG and some NEF previews are held. + pub const STRIP_OFFSETS: u16 = 0x0111; + pub const STRIP_BYTE_COUNTS: u16 = 0x0117; + pub const IMAGE_WIDTH: u16 = 0x0100; + pub const IMAGE_LENGTH: u16 = 0x0101; + /// 1 = full-resolution sensor data, 0 = a reduced-resolution preview. + pub const NEW_SUBFILE_TYPE: u16 = 0x00FE; + pub const COMPRESSION: u16 = 0x0103; + pub const SUB_IFDS: u16 = 0x014A; +} + +/// JPEG compression, as opposed to raw sensor data. +const COMPRESSION_JPEG: u32 = 6; +const COMPRESSION_OLD_JPEG: u32 = 7; + +fn preview_from_entries( + tiff: &TiffReader, + entries: &[Entry], + file_len: u64, +) -> Option { + let get = |t: u16| entries.iter().find(|e| e.tag == t); + + // Reject the full-resolution image: it is sensor data, not a preview, and + // "locating" it would transfer the whole file — the exact cost this avoids. + if let Some(e) = get(tag::NEW_SUBFILE_TYPE) { + if tiff.scalar(e)? == 0 && get(tag::JPEG_INTERCHANGE_FORMAT).is_none() { + // Subfile type 0 means full resolution. Only continue if it is a + // JPEG interchange entry, which a main image never is. + return None; + } + } + + // Strip-based entries must be JPEG-compressed; an uncompressed strip is + // raw sensor data that no JPEG decoder will read. + let (offset, length) = if let (Some(o), Some(l)) = ( + get(tag::JPEG_INTERCHANGE_FORMAT), + get(tag::JPEG_INTERCHANGE_FORMAT_LENGTH), + ) { + (tiff.scalar(o)? as u64, tiff.scalar(l)? as u64) + } else if let (Some(o), Some(l), Some(c)) = ( + get(tag::STRIP_OFFSETS), + get(tag::STRIP_BYTE_COUNTS), + get(tag::COMPRESSION), + ) { + let compression = tiff.scalar(c)?; + if compression != COMPRESSION_JPEG && compression != COMPRESSION_OLD_JPEG { + return None; + } + // A multi-strip image is tiled sensor data, not a single JPEG. + if o.count != 1 || l.count != 1 { + return None; + } + (tiff.scalar(o)? as u64, tiff.scalar(l)? as u64) + } else { + return None; + }; + + // Everything below is validation against a hostile file (NFR-SEC-1). + if length == 0 { + return None; + } + let end = offset.checked_add(length)?; + if end > file_len { + return None; + } + // A "preview" the size of the whole file is the full image mislabelled. + if length > file_len / 2 { + return None; + } + + Some(PreviewLocation { + range: offset..end, + width: get(tag::IMAGE_WIDTH).and_then(|e| tiff.scalar(e)), + height: get(tag::IMAGE_LENGTH).and_then(|e| tiff.scalar(e)), + }) +} + +/// One IFD entry. +#[derive(Debug, Clone, Copy)] +struct Entry { + tag: u16, + kind: u16, + count: u32, + /// The raw 4-byte value field — either the value itself or an offset to it. + value: u32, +} + +/// A minimal TIFF structure reader. +/// +/// Deliberately not a general TIFF parser: it reads the IFD chain and entry +/// values and nothing else, because that is all locating a preview needs. +struct TiffReader<'a> { + data: &'a [u8], + little_endian: bool, + first_ifd: u32, +} + +impl<'a> TiffReader<'a> { + fn new(data: &'a [u8]) -> Option { + if data.len() < 8 { + return None; + } + let little_endian = match &data[0..2] { + b"II" => true, + b"MM" => false, + _ => return None, + }; + + let magic = read_u16(data, 2, little_endian)?; + // 42 is TIFF; 0x4F52 and 0x5352 are ORF's variants, which are + // otherwise TIFF-shaped. + if magic != 42 && magic != 0x4F52 && magic != 0x5352 { + return None; + } + + let first_ifd = read_u32(data, 4, little_endian)?; + Some(Self { + data, + little_endian, + first_ifd, + }) + } + + /// Every IFD worth searching: the chain from the header, plus any SubIFDs. + /// + /// Bounded, because a malformed file can point an IFD at itself and a + /// naive walk would never terminate. + fn ifd_offsets(&self) -> Vec { + const MAX_IFDS: usize = 16; + let mut out = Vec::new(); + let mut seen = std::collections::HashSet::new(); + let mut next = self.first_ifd; + + while next != 0 && out.len() < MAX_IFDS && seen.insert(next) { + out.push(next); + + // SubIFDs hold the preview in DNG and several NEF variants. + if let Some(entries) = self.read_ifd(next) { + if let Some(sub) = entries.iter().find(|e| e.tag == tag::SUB_IFDS) { + for offset in self.offsets(sub) { + if out.len() < MAX_IFDS && seen.insert(offset) { + out.push(offset); + } + } + } + } + + match self.next_ifd_offset(next) { + Some(n) => next = n, + None => break, + } + } + out + } + + fn read_ifd(&self, offset: u32) -> Option> { + let base = offset as usize; + let count = read_u16(self.data, base, self.little_endian)? as usize; + + // A plausible IFD has tens of entries, not thousands. A huge count is + // a corrupt or hostile file, and allocating for it is the bug. + if count > 512 { + return None; + } + + let mut entries = Vec::with_capacity(count); + for i in 0..count { + let e = base + 2 + i * 12; + entries.push(Entry { + tag: read_u16(self.data, e, self.little_endian)?, + kind: read_u16(self.data, e + 2, self.little_endian)?, + count: read_u32(self.data, e + 4, self.little_endian)?, + value: read_u32(self.data, e + 8, self.little_endian)?, + }); + } + Some(entries) + } + + fn next_ifd_offset(&self, ifd: u32) -> Option { + let base = ifd as usize; + let count = read_u16(self.data, base, self.little_endian)? as usize; + read_u32(self.data, base + 2 + count * 12, self.little_endian) + } + + /// An entry's value as a single number. + /// + /// Handles the inline case only for the scalar types a preview entry uses; + /// anything larger than four bytes is stored out of line and read through + /// its offset. + fn scalar(&self, e: &Entry) -> Option { + match e.kind { + // SHORT, inline when count is 1. + 3 if e.count == 1 => Some(if self.little_endian { + e.value & 0xFFFF + } else { + // Big-endian packs a short into the high half of the field. + e.value >> 16 + }), + // LONG, always inline at count 1. + 4 if e.count == 1 => Some(e.value), + // A count above one points elsewhere; take the first element. + 3 => read_u16(self.data, e.value as usize, self.little_endian).map(u32::from), + 4 => read_u32(self.data, e.value as usize, self.little_endian), + _ => None, + } + } + + /// An ASCII entry's string value. + /// + /// EXIF strings are NUL-terminated and often padded, and camera vendors + /// pad with spaces too — both are trimmed, since a model name with a + /// trailing NUL compares unequal to the same name without one. + fn ascii(&self, e: &Entry) -> Option { + // Type 2 is ASCII. Up to four bytes live inline; longer strings are + // stored at the offset in the value field. + if e.kind != 2 || e.count == 0 { + return None; + } + let len = e.count as usize; + let bytes = if len <= 4 { + let raw = if self.little_endian { + e.value.to_le_bytes() + } else { + e.value.to_be_bytes() + }; + raw[..len.min(4)].to_vec() + } else { + self.data.get(e.value as usize..e.value as usize + len)?.to_vec() + }; + + let s = String::from_utf8_lossy(&bytes); + let s = s.trim_end_matches('\0').trim(); + if s.is_empty() { + None + } else { + Some(s.to_string()) + } + } + + /// An entry's values as a list of offsets (for SubIFDs). + fn offsets(&self, e: &Entry) -> Vec { + if e.kind != 4 { + return Vec::new(); + } + if e.count == 1 { + return vec![e.value]; + } + // Bounded: a SubIFD list is a handful of entries, never thousands. + (0..e.count.min(8)) + .filter_map(|i| { + read_u32( + self.data, + e.value as usize + (i as usize) * 4, + self.little_endian, + ) + }) + .collect() + } +} + +fn read_u16(data: &[u8], at: usize, le: bool) -> Option { + let b = data.get(at..at + 2)?; + Some(if le { + u16::from_le_bytes([b[0], b[1]]) + } else { + u16::from_be_bytes([b[0], b[1]]) + }) +} + +fn read_u32(data: &[u8], at: usize, le: bool) -> Option { + let b = data.get(at..at + 4)?; + Some(if le { + u32::from_le_bytes([b[0], b[1], b[2], b[3]]) + } else { + u32::from_be_bytes([b[0], b[1], b[2], b[3]]) + }) +} + +/// Read EXIF from a JPEG's APP1 segment. +/// +/// A JPEG's EXIF block is a complete TIFF structure embedded in an `APP1` +/// marker, so the reader above does the work — only finding the block differs. +/// +/// This exists because rawler decodes no JPEG at all, and a photo library is +/// full of them: camera JPEGs, and in this project's reference library nearly +/// six thousand scanned frames. Without it every one is undated and missing +/// from the timeline. +pub fn jpeg_metadata(bytes: &[u8]) -> Result { + let tiff_start = find_exif_tiff(bytes) + .ok_or_else(|| crate::DecodeError::Metadata("no EXIF segment".into()))?; + tiff_metadata(&bytes[tiff_start..]) +} + +/// Read EXIF from a bare TIFF structure. +/// +/// Serves two callers: a JPEG's APP1 payload, and a TIFF-derived RAW whose +/// primary decoder returned no date. The second case is real — rawler reports +/// no `DateTimeOriginal` for some DNGs whose tag sits plainly at byte 826 — +/// and without this fallback those images are silently undated. +pub fn tiff_metadata(tiff_data: &[u8]) -> Result { + let reader = TiffReader::new(tiff_data) + .ok_or_else(|| crate::DecodeError::Metadata("malformed EXIF header".into()))?; + + let mut md = crate::Metadata::default(); + // Fallback dates accumulate across *all* IFDs before being resolved. They + // must not be settled per-IFD: one reference scanner writes `DateTime` in + // the main IFD and `DateTimeDigitized` in the Exif sub-IFD, 102 seconds + // apart, so resolving after the first would take the worse of the two. + let mut fb = FallbackDates::default(); + + for ifd in reader.ifd_offsets() { + let Some(entries) = reader.read_ifd(ifd) else { + continue; + }; + read_exif_entries(&reader, &entries, &mut md, &mut fb); + + // The interesting tags live in the Exif sub-IFD, which the main IFD + // points at rather than containing. + if let Some(e) = entries.iter().find(|e| e.tag == EXIF_IFD_POINTER) { + if let Some(sub) = reader.scalar(e).and_then(|o| reader.read_ifd(o)) { + read_exif_entries(&reader, &sub, &mut md, &mut fb); + } + } + } + + // Ranked: when the shutter fired, else when the image was digitised, else + // when the file was last written. `DateTime` moves on every re-save, so it + // is the last resort rather than the first match. + if md.captured_at.is_none() { + md.captured_at = fb.digitized.or(fb.modified); + } + Ok(md) +} + +/// Offset of the TIFF header inside a JPEG's `APP1` EXIF segment. +/// +/// Walks the marker chain rather than scanning for the `Exif\0\0` magic: +/// scanning could match those bytes inside compressed image data and point the +/// TIFF reader at noise. +fn find_exif_tiff(bytes: &[u8]) -> Option { + if !bytes.starts_with(&[0xFF, 0xD8]) { + return None; + } + let mut i = 2; + // Bounded by the header slice callers pass; a malformed length field + // cannot walk past the end because every read is checked. + while i + 4 <= bytes.len() { + if bytes[i] != 0xFF { + return None; + } + let marker = bytes[i + 1]; + // Start of scan: image data follows, and no more headers. + if marker == 0xDA { + return None; + } + let len = u16::from_be_bytes([bytes[i + 2], bytes[i + 3]]) as usize; + if len < 2 { + return None; + } + // APP1 carrying the "Exif\0\0" identifier. + if marker == 0xE1 { + let seg = bytes.get(i + 4..i + 2 + len)?; + if seg.starts_with(b"Exif\0\0") { + return Some(i + 4 + 6); + } + } + i += 2 + len; + } + None +} + +/// Exif sub-IFD pointer, where the capture tags actually live. +const EXIF_IFD_POINTER: u16 = 0x8769; + +mod exif_tag { + pub const MAKE: u16 = 0x010F; + pub const MODEL: u16 = 0x0110; + /// When the shutter fired. Absent on scanner output. + pub const DATE_TIME_ORIGINAL: u16 = 0x9003; + /// When the file was written. A camera sets both; a **scanner sets only + /// this one**, so without it every scanned frame is undated — 5,712 of + /// them in this project's reference library. + pub const DATE_TIME: u16 = 0x0132; + /// Digitisation time. Another fallback some devices fill instead. + pub const DATE_TIME_DIGITIZED: u16 = 0x9004; + pub const OFFSET_TIME_ORIGINAL: u16 = 0x9011; + pub const ISO: u16 = 0x8827; + pub const LENS_MODEL: u16 = 0xA434; + pub const PIXEL_X: u16 = 0xA002; + pub const PIXEL_Y: u16 = 0xA003; +} + +/// Dates that stand in for a missing `DateTimeOriginal`. +/// +/// Collected across every IFD and ranked once at the end, because the two can +/// live in different IFDs and disagree. +#[derive(Default)] +struct FallbackDates { + /// When the image was digitised. A scanner's real capture time. + digitized: Option, + /// When the file was last written. Moves on re-save, so lowest rank. + modified: Option, +} + +fn read_exif_entries( + r: &TiffReader, + entries: &[Entry], + md: &mut crate::Metadata, + fb: &mut FallbackDates, +) { + + for e in entries { + match e.tag { + exif_tag::MAKE => md.make = r.ascii(e), + exif_tag::MODEL => md.model = r.ascii(e), + exif_tag::LENS_MODEL => md.lens = r.ascii(e), + exif_tag::ISO => md.iso = r.scalar(e), + exif_tag::PIXEL_X => md.width = r.scalar(e), + exif_tag::PIXEL_Y => md.height = r.scalar(e), + exif_tag::DATE_TIME_ORIGINAL => { + if let Some(t) = r.ascii(e).as_deref().and_then(crate::parse_exif_datetime) { + md.captured_at = Some(t); + } + } + exif_tag::DATE_TIME_DIGITIZED => { + if let Some(t) = r.ascii(e).as_deref().and_then(crate::parse_exif_datetime) { + fb.digitized.get_or_insert(t); + } + } + exif_tag::DATE_TIME => { + if let Some(t) = r.ascii(e).as_deref().and_then(crate::parse_exif_datetime) { + fb.modified.get_or_insert(t); + } + } + exif_tag::OFFSET_TIME_ORIGINAL => { + md.captured_offset = r.ascii(e).as_deref().and_then(crate::parse_exif_offset) + } + _ => {} + } + } +} + +/// Whether a byte slice is a complete JPEG. +/// +/// A truncated JPEG decodes to a partial image rather than an error — the +/// exact failure that made range-fetched thumbnails render as the top tenth of +/// the frame. Checking for the end-of-image marker catches it before the +/// result reaches a cache or a screen. +pub fn is_complete_jpeg(bytes: &[u8]) -> bool { + bytes.len() > 4 + && bytes.starts_with(&[0xFF, 0xD8]) + // Trailing padding after EOI is legal and does occur, so scan the tail + // rather than testing only the final two bytes. + && bytes + .rchunks(64) + .next() + .map(|tail| tail.windows(2).any(|w| w == [0xFF, 0xD9])) + .unwrap_or(false) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Build a little-endian TIFF header with one IFD. + fn tiff(entries: &[(u16, u16, u32, u32)], next_ifd: u32) -> Vec { + let mut v = Vec::new(); + v.extend_from_slice(b"II"); + v.extend_from_slice(&42u16.to_le_bytes()); + v.extend_from_slice(&8u32.to_le_bytes()); // first IFD at offset 8 + + v.extend_from_slice(&(entries.len() as u16).to_le_bytes()); + for (tag, kind, count, value) in entries { + v.extend_from_slice(&tag.to_le_bytes()); + v.extend_from_slice(&kind.to_le_bytes()); + v.extend_from_slice(&count.to_le_bytes()); + v.extend_from_slice(&value.to_le_bytes()); + } + v.extend_from_slice(&next_ifd.to_le_bytes()); + v.resize(v.len().max(1024), 0); + v + } + + #[test] + fn finds_a_jpeg_interchange_preview() { + let h = tiff( + &[ + (tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 100_000), + (tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 1_500_000), + (tag::IMAGE_WIDTH, 4, 1, 1620), + (tag::IMAGE_LENGTH, 4, 1, 1080), + ], + 0, + ); + let loc = locate_preview(&h, 25_000_000).expect("a preview"); + assert_eq!(loc.range, 100_000..1_600_000); + assert_eq!(loc.width, Some(1620)); + assert_eq!(loc.height, Some(1080)); + } + + #[test] + fn a_range_past_the_end_of_file_is_rejected() { + // The check that stops a corrupt offset becoming a wild range request. + let h = tiff( + &[ + (tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 20_000_000), + (tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 10_000_000), + ], + 0, + ); + assert!(locate_preview(&h, 25_000_000).is_none()); + } + + #[test] + fn a_preview_larger_than_half_the_file_is_rejected() { + // That is the full image mislabelled; "locating" it would transfer the + // whole file, which is what this exists to avoid. + let h = tiff( + &[ + (tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 100), + (tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 9_000_000), + ], + 0, + ); + assert!(locate_preview(&h, 10_000_000).is_none()); + } + + #[test] + fn a_zero_length_preview_is_rejected() { + let h = tiff( + &[ + (tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 100), + (tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 0), + ], + 0, + ); + assert!(locate_preview(&h, 1_000_000).is_none()); + } + + #[test] + fn uncompressed_strips_are_not_mistaken_for_a_preview() { + // Raw sensor data lives in strips too; feeding it to a JPEG decoder + // would produce noise. + let h = tiff( + &[ + (tag::STRIP_OFFSETS, 4, 1, 50_000), + (tag::STRIP_BYTE_COUNTS, 4, 1, 800_000), + (tag::COMPRESSION, 3, 1, 1), // uncompressed + ], + 0, + ); + assert!(locate_preview(&h, 25_000_000).is_none()); + } + + #[test] + fn jpeg_compressed_strips_are_accepted() { + let h = tiff( + &[ + (tag::STRIP_OFFSETS, 4, 1, 50_000), + (tag::STRIP_BYTE_COUNTS, 4, 1, 800_000), + (tag::COMPRESSION, 3, 1, COMPRESSION_JPEG), + ], + 0, + ); + let loc = locate_preview(&h, 25_000_000).expect("a preview"); + assert_eq!(loc.range, 50_000..850_000); + } + + #[test] + fn multi_strip_images_are_rejected() { + // Several strips means tiled sensor data, not one contiguous JPEG. + let h = tiff( + &[ + (tag::STRIP_OFFSETS, 4, 8, 50_000), + (tag::STRIP_BYTE_COUNTS, 4, 8, 800_000), + (tag::COMPRESSION, 3, 1, COMPRESSION_JPEG), + ], + 0, + ); + assert!(locate_preview(&h, 25_000_000).is_none()); + } + + #[test] + fn big_endian_files_parse() { + // Nikon and Olympus ship big-endian containers. + let mut v = Vec::new(); + v.extend_from_slice(b"MM"); + v.extend_from_slice(&42u16.to_be_bytes()); + v.extend_from_slice(&8u32.to_be_bytes()); + v.extend_from_slice(&2u16.to_be_bytes()); + for (tag, kind, count, value) in [ + (tag::JPEG_INTERCHANGE_FORMAT, 4u16, 1u32, 4096u32), + (tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 900_000), + ] { + v.extend_from_slice(&tag.to_be_bytes()); + v.extend_from_slice(&kind.to_be_bytes()); + v.extend_from_slice(&count.to_be_bytes()); + v.extend_from_slice(&value.to_be_bytes()); + } + v.extend_from_slice(&0u32.to_be_bytes()); + v.resize(1024, 0); + + let loc = locate_preview(&v, 25_000_000).expect("a preview"); + assert_eq!(loc.range, 4096..904_096); + } + + #[test] + fn a_big_endian_short_reads_from_the_high_half() { + // The classic TIFF trap: a SHORT is left-justified in the 4-byte value + // field on big-endian, so reading it as a LONG yields a huge number. + let mut v = Vec::new(); + v.extend_from_slice(b"MM"); + v.extend_from_slice(&42u16.to_be_bytes()); + v.extend_from_slice(&8u32.to_be_bytes()); + v.extend_from_slice(&4u16.to_be_bytes()); + for (tag, kind, count, value) in [ + (tag::STRIP_OFFSETS, 4u16, 1u32, 1000u32), + (tag::STRIP_BYTE_COUNTS, 4, 1, 500_000), + (tag::COMPRESSION, 3, 1, (COMPRESSION_JPEG) << 16), + (tag::IMAGE_WIDTH, 3, 1, 1620u32 << 16), + ] { + v.extend_from_slice(&tag.to_be_bytes()); + v.extend_from_slice(&kind.to_be_bytes()); + v.extend_from_slice(&count.to_be_bytes()); + v.extend_from_slice(&value.to_be_bytes()); + } + v.extend_from_slice(&0u32.to_be_bytes()); + v.resize(1024, 0); + + let loc = locate_preview(&v, 25_000_000).expect("a preview"); + assert_eq!(loc.width, Some(1620), "short read from the wrong half"); + } + + #[test] + fn the_largest_preview_wins_across_ifds() { + // Cameras carry both a tiny thumbnail and a screen-sized preview; the + // larger downscales better. + let mut v = tiff( + &[ + (tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 1000), + (tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 8_000), + ], + 200, + ); + // A second IFD at offset 200 with a much larger preview. + let second = 200usize; + v[second..second + 2].copy_from_slice(&2u16.to_le_bytes()); + for (i, (tag, kind, count, value)) in [ + (tag::JPEG_INTERCHANGE_FORMAT, 4u16, 1u32, 20_000u32), + (tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 1_200_000), + ] + .iter() + .enumerate() + { + let e = second + 2 + i * 12; + v[e..e + 2].copy_from_slice(&tag.to_le_bytes()); + v[e + 2..e + 4].copy_from_slice(&kind.to_le_bytes()); + v[e + 4..e + 8].copy_from_slice(&count.to_le_bytes()); + v[e + 8..e + 12].copy_from_slice(&value.to_le_bytes()); + } + + let loc = locate_preview(&v, 25_000_000).expect("a preview"); + assert_eq!(loc.len(), 1_200_000, "should pick the larger"); + } + + #[test] + fn a_self_referential_ifd_chain_terminates() { + // A malformed file pointing an IFD at itself must not hang the app. + let h = tiff(&[(tag::IMAGE_WIDTH, 4, 1, 100)], 8); + let _ = locate_preview(&h, 1_000_000); + } + + #[test] + fn an_absurd_entry_count_is_rejected_not_allocated() { + let mut v = Vec::new(); + v.extend_from_slice(b"II"); + v.extend_from_slice(&42u16.to_le_bytes()); + v.extend_from_slice(&8u32.to_le_bytes()); + v.extend_from_slice(&60000u16.to_le_bytes()); // claims 60k entries + v.resize(1024, 0); + assert!(locate_preview(&v, 1_000_000).is_none()); + } + + #[test] + fn non_tiff_input_declines_cleanly() { + assert!(locate_preview(b"not a tiff at all", 1000).is_none()); + assert!(locate_preview(&[], 1000).is_none()); + // CR3 is ISO-BMFF, not TIFF — declining is correct. + assert!(locate_preview(b"\0\0\0\x18ftypcrx ", 1000).is_none()); + } + + + /// Build a JPEG carrying an APP1 EXIF block with the given IFD entries. + fn jpeg_with_exif(entries: &[(u16, u16, u32, u32)], extra: &[u8]) -> Vec { + let mut tiff = Vec::new(); + tiff.extend_from_slice(b"II"); + tiff.extend_from_slice(&42u16.to_le_bytes()); + tiff.extend_from_slice(&8u32.to_le_bytes()); + tiff.extend_from_slice(&(entries.len() as u16).to_le_bytes()); + for (tag, kind, count, value) in entries { + tiff.extend_from_slice(&tag.to_le_bytes()); + tiff.extend_from_slice(&kind.to_le_bytes()); + tiff.extend_from_slice(&count.to_le_bytes()); + tiff.extend_from_slice(&value.to_le_bytes()); + } + tiff.extend_from_slice(&0u32.to_le_bytes()); + tiff.extend_from_slice(extra); + + let payload_len = (tiff.len() + 6 + 2) as u16; + let mut out = vec![0xFF, 0xD8, 0xFF, 0xE1]; + out.extend_from_slice(&payload_len.to_be_bytes()); + out.extend_from_slice(b"Exif\0\0"); + out.extend_from_slice(&tiff); + out + } + + #[test] + fn jpeg_exif_yields_a_capture_time() { + // rawler decodes no JPEG at all, so without this path every JPEG in a + // library is undated — 5,712 scanned frames in the reference library. + let date = b"2013:06:28 23:32:54\0"; + let mut extra = Vec::new(); + let date_offset = 8 + 2 + 12 + 4; + extra.extend_from_slice(date); + + let jpeg = jpeg_with_exif( + &[( + exif_tag::DATE_TIME_ORIGINAL, + 2, + date.len() as u32, + date_offset as u32, + )], + &extra, + ); + + let md = jpeg_metadata(&jpeg).expect("EXIF"); + assert_eq!(md.captured_at, Some(1_372_462_374)); + } + + #[test] + fn a_jpeg_without_exif_reports_no_segment() { + assert!(jpeg_metadata(&[0xFF, 0xD8, 0xFF, 0xDA, 0, 2]).is_err()); + assert!(jpeg_metadata(b"not a jpeg").is_err()); + } + + #[test] + fn ascii_values_lose_their_nul_padding() { + // A model name with a trailing NUL compares unequal to the same name + // without one, which would split one camera into two in any grouping. + let model = b"CanoScan 9000F Mark II\0"; + let mut extra = Vec::new(); + let off = 8 + 2 + 12 + 4; + extra.extend_from_slice(model); + + let jpeg = jpeg_with_exif( + &[(exif_tag::MODEL, 2, model.len() as u32, off as u32)], + &extra, + ); + let md = jpeg_metadata(&jpeg).expect("EXIF"); + assert_eq!(md.model.as_deref(), Some("CanoScan 9000F Mark II")); + } + + #[test] + fn a_marker_walk_does_not_run_off_a_truncated_file() { + // Untrusted input (NFR-SEC-1): a length field claiming more than the + // file holds must not read past the end. + let mut jpeg = vec![0xFF, 0xD8, 0xFF, 0xE1]; + jpeg.extend_from_slice(&60000u16.to_be_bytes()); + jpeg.extend_from_slice(b"Exif\0\0"); + assert!(jpeg_metadata(&jpeg).is_err()); + } + + #[test] + fn truncated_jpegs_are_detected() { + // The bug this whole module exists to fix: a short read decodes to a + // partial image rather than failing, so it must be caught by + // inspection. + let mut complete = vec![0xFF, 0xD8]; + complete.extend_from_slice(&[0x00; 200]); + complete.extend_from_slice(&[0xFF, 0xD9]); + assert!(is_complete_jpeg(&complete)); + + let truncated = &complete[..complete.len() - 2]; + assert!(!is_complete_jpeg(truncated)); + } + + #[test] + fn non_jpeg_bytes_are_not_complete_jpegs() { + assert!(!is_complete_jpeg(&[])); + assert!(!is_complete_jpeg(&[0xFF, 0xD9])); + assert!(!is_complete_jpeg(b"PNG\r\n")); + } +}