//! TRACES: FR-NC-3 | FR-CULL-2 //! Finding an embedded preview's byte range from a file header alone. //! //! # Why this exists //! //! Remote browsing must not transfer whole RAW files (FR-NC-3). The obvious //! shortcut — fetch a fixed prefix and hope the preview is inside it — does //! not work: an embedded JPEG typically starts a few hundred KB in and runs //! for one to three MB, so a truncated fetch yields a JPEG whose scanlines //! stop partway down. Decoders render what they have rather than erroring, so //! the failure looks like a corrupt image, not a short read. //! //! What FR-NC-3 actually specifies is two-stage: read the header, *parse the //! container* to locate the preview, then fetch exactly those bytes. //! //! # Scope //! //! This reads TIFF-structured containers — CR2, NEF, ARW, DNG, and ORF all //! carry their previews in IFD entries. CR3 is ISO-BMFF and is not handled //! here; it falls back to the caller's whole-file path, which is correct if //! slower. A locator that returned a wrong range would be far worse than one //! that declines. //! //! Every offset read from the file is treated as hostile (NFR-SEC-1): bounds //! are checked against the real file length, never trusted. use std::ops::Range; /// Where a preview lives inside its container. #[derive(Debug, Clone, PartialEq, Eq)] pub struct PreviewLocation { /// Byte range of the JPEG, ready to hand to a `Range:` request. pub range: Range, /// Pixel dimensions where the container declared them. Used to pick the /// largest preview that is still smaller than a full decode. pub width: Option, pub height: Option, } impl PreviewLocation { pub fn len(&self) -> u64 { self.range.end - self.range.start } pub fn is_empty(&self) -> bool { self.range.end <= self.range.start } } /// How many bytes of header a caller should fetch before calling this. /// /// Large enough to cover the IFD chain in the formats measured, small enough /// that a miss costs little. Metadata alone needs less, but IFD1/IFD2 entries /// for the preview sit further in on some bodies. pub const HEADER_BYTES: u64 = 256 * 1024; /// Locate the largest embedded preview at or below `max_edge`, if any. /// /// `header` is the first [`HEADER_BYTES`] of the file; `file_len` is the whole /// file's length, needed to reject offsets that point past the end. /// /// Returns `None` when the container is not TIFF-structured, declares no /// preview, or declares one whose range is not credible. pub fn locate_preview(header: &[u8], file_len: u64) -> Option { let tiff = TiffReader::new(header)?; let mut best: Option = None; for ifd_offset in tiff.ifd_offsets() { let Some(entries) = tiff.read_ifd(ifd_offset) else { continue; }; if let Some(loc) = preview_from_entries(&tiff, &entries, file_len) { // Prefer the largest, since a bigger preview downscales better — // but anything is better than nothing. let better = match (&best, &loc) { (None, _) => true, (Some(b), l) => l.len() > b.len(), }; if better { best = Some(loc); } } } best } /// TIFF tags that carry a preview's location. mod tag { /// Legacy thumbnail offset/length (IFD1 in most makes). pub const JPEG_INTERCHANGE_FORMAT: u16 = 0x0201; pub const JPEG_INTERCHANGE_FORMAT_LENGTH: u16 = 0x0202; /// Strip-based storage, which is how DNG and some NEF previews are held. pub const STRIP_OFFSETS: u16 = 0x0111; pub const STRIP_BYTE_COUNTS: u16 = 0x0117; pub const IMAGE_WIDTH: u16 = 0x0100; pub const IMAGE_LENGTH: u16 = 0x0101; /// 1 = full-resolution sensor data, 0 = a reduced-resolution preview. pub const NEW_SUBFILE_TYPE: u16 = 0x00FE; pub const COMPRESSION: u16 = 0x0103; pub const SUB_IFDS: u16 = 0x014A; } /// JPEG compression, as opposed to raw sensor data. const COMPRESSION_JPEG: u32 = 6; const COMPRESSION_OLD_JPEG: u32 = 7; fn preview_from_entries( tiff: &TiffReader, entries: &[Entry], file_len: u64, ) -> Option { let get = |t: u16| entries.iter().find(|e| e.tag == t); // Reject the full-resolution image: it is sensor data, not a preview, and // "locating" it would transfer the whole file — the exact cost this avoids. if let Some(e) = get(tag::NEW_SUBFILE_TYPE) { if tiff.scalar(e)? == 0 && get(tag::JPEG_INTERCHANGE_FORMAT).is_none() { // Subfile type 0 means full resolution. Only continue if it is a // JPEG interchange entry, which a main image never is. return None; } } // Strip-based entries must be JPEG-compressed; an uncompressed strip is // raw sensor data that no JPEG decoder will read. let (offset, length) = if let (Some(o), Some(l)) = ( get(tag::JPEG_INTERCHANGE_FORMAT), get(tag::JPEG_INTERCHANGE_FORMAT_LENGTH), ) { (tiff.scalar(o)? as u64, tiff.scalar(l)? as u64) } else if let (Some(o), Some(l), Some(c)) = ( get(tag::STRIP_OFFSETS), get(tag::STRIP_BYTE_COUNTS), get(tag::COMPRESSION), ) { let compression = tiff.scalar(c)?; if compression != COMPRESSION_JPEG && compression != COMPRESSION_OLD_JPEG { return None; } // A multi-strip image is tiled sensor data, not a single JPEG. if o.count != 1 || l.count != 1 { return None; } (tiff.scalar(o)? as u64, tiff.scalar(l)? as u64) } else { return None; }; // Everything below is validation against a hostile file (NFR-SEC-1). if length == 0 { return None; } let end = offset.checked_add(length)?; if end > file_len { return None; } // A "preview" the size of the whole file is the full image mislabelled. if length > file_len / 2 { return None; } Some(PreviewLocation { range: offset..end, width: get(tag::IMAGE_WIDTH).and_then(|e| tiff.scalar(e)), height: get(tag::IMAGE_LENGTH).and_then(|e| tiff.scalar(e)), }) } /// One IFD entry. #[derive(Debug, Clone, Copy)] struct Entry { tag: u16, kind: u16, count: u32, /// The raw 4-byte value field — either the value itself or an offset to it. value: u32, } /// A minimal TIFF structure reader. /// /// Deliberately not a general TIFF parser: it reads the IFD chain and entry /// values and nothing else, because that is all locating a preview needs. struct TiffReader<'a> { data: &'a [u8], little_endian: bool, first_ifd: u32, } impl<'a> TiffReader<'a> { fn new(data: &'a [u8]) -> Option { if data.len() < 8 { return None; } let little_endian = match &data[0..2] { b"II" => true, b"MM" => false, _ => return None, }; let magic = read_u16(data, 2, little_endian)?; // 42 is TIFF; 0x4F52 and 0x5352 are ORF's variants, which are // otherwise TIFF-shaped. if magic != 42 && magic != 0x4F52 && magic != 0x5352 { return None; } let first_ifd = read_u32(data, 4, little_endian)?; Some(Self { data, little_endian, first_ifd, }) } /// Every IFD worth searching: the chain from the header, plus any SubIFDs. /// /// Bounded, because a malformed file can point an IFD at itself and a /// naive walk would never terminate. fn ifd_offsets(&self) -> Vec { const MAX_IFDS: usize = 16; let mut out = Vec::new(); let mut seen = std::collections::HashSet::new(); let mut next = self.first_ifd; while next != 0 && out.len() < MAX_IFDS && seen.insert(next) { out.push(next); // SubIFDs hold the preview in DNG and several NEF variants. if let Some(entries) = self.read_ifd(next) { if let Some(sub) = entries.iter().find(|e| e.tag == tag::SUB_IFDS) { for offset in self.offsets(sub) { if out.len() < MAX_IFDS && seen.insert(offset) { out.push(offset); } } } } match self.next_ifd_offset(next) { Some(n) => next = n, None => break, } } out } fn read_ifd(&self, offset: u32) -> Option> { let base = offset as usize; let count = read_u16(self.data, base, self.little_endian)? as usize; // A plausible IFD has tens of entries, not thousands. A huge count is // a corrupt or hostile file, and allocating for it is the bug. if count > 512 { return None; } let mut entries = Vec::with_capacity(count); for i in 0..count { let e = base + 2 + i * 12; entries.push(Entry { tag: read_u16(self.data, e, self.little_endian)?, kind: read_u16(self.data, e + 2, self.little_endian)?, count: read_u32(self.data, e + 4, self.little_endian)?, value: read_u32(self.data, e + 8, self.little_endian)?, }); } Some(entries) } fn next_ifd_offset(&self, ifd: u32) -> Option { let base = ifd as usize; let count = read_u16(self.data, base, self.little_endian)? as usize; read_u32(self.data, base + 2 + count * 12, self.little_endian) } /// An entry's value as a single number. /// /// Handles the inline case only for the scalar types a preview entry uses; /// anything larger than four bytes is stored out of line and read through /// its offset. fn scalar(&self, e: &Entry) -> Option { match e.kind { // SHORT, inline when count is 1. 3 if e.count == 1 => Some(if self.little_endian { e.value & 0xFFFF } else { // Big-endian packs a short into the high half of the field. e.value >> 16 }), // LONG, always inline at count 1. 4 if e.count == 1 => Some(e.value), // A count above one points elsewhere; take the first element. 3 => read_u16(self.data, e.value as usize, self.little_endian).map(u32::from), 4 => read_u32(self.data, e.value as usize, self.little_endian), _ => None, } } /// An ASCII entry's string value. /// /// EXIF strings are NUL-terminated and often padded, and camera vendors /// pad with spaces too — both are trimmed, since a model name with a /// trailing NUL compares unequal to the same name without one. fn ascii(&self, e: &Entry) -> Option { // Type 2 is ASCII. Up to four bytes live inline; longer strings are // stored at the offset in the value field. if e.kind != 2 || e.count == 0 { return None; } let len = e.count as usize; let bytes = if len <= 4 { let raw = if self.little_endian { e.value.to_le_bytes() } else { e.value.to_be_bytes() }; raw[..len.min(4)].to_vec() } else { self.data .get(e.value as usize..e.value as usize + len)? .to_vec() }; let s = String::from_utf8_lossy(&bytes); let s = s.trim_end_matches('\0').trim(); if s.is_empty() { None } else { Some(s.to_string()) } } /// An entry's values as a list of offsets (for SubIFDs). fn offsets(&self, e: &Entry) -> Vec { if e.kind != 4 { return Vec::new(); } if e.count == 1 { return vec![e.value]; } // Bounded: a SubIFD list is a handful of entries, never thousands. (0..e.count.min(8)) .filter_map(|i| { read_u32( self.data, e.value as usize + (i as usize) * 4, self.little_endian, ) }) .collect() } } fn read_u16(data: &[u8], at: usize, le: bool) -> Option { let b = data.get(at..at + 2)?; Some(if le { u16::from_le_bytes([b[0], b[1]]) } else { u16::from_be_bytes([b[0], b[1]]) }) } fn read_u32(data: &[u8], at: usize, le: bool) -> Option { let b = data.get(at..at + 4)?; Some(if le { u32::from_le_bytes([b[0], b[1], b[2], b[3]]) } else { u32::from_be_bytes([b[0], b[1], b[2], b[3]]) }) } /// Read EXIF from a JPEG's APP1 segment. /// /// A JPEG's EXIF block is a complete TIFF structure embedded in an `APP1` /// marker, so the reader above does the work — only finding the block differs. /// /// This exists because rawler decodes no JPEG at all, and a photo library is /// full of them: camera JPEGs, and in this project's reference library nearly /// six thousand scanned frames. Without it every one is undated and missing /// from the timeline. pub fn jpeg_metadata(bytes: &[u8]) -> Result { let tiff_start = find_exif_tiff(bytes) .ok_or_else(|| crate::DecodeError::Metadata("no EXIF segment".into()))?; tiff_metadata(&bytes[tiff_start..]) } /// Read EXIF from a bare TIFF structure. /// /// Serves two callers: a JPEG's APP1 payload, and a TIFF-derived RAW whose /// primary decoder returned no date. The second case is real — rawler reports /// no `DateTimeOriginal` for some DNGs whose tag sits plainly at byte 826 — /// and without this fallback those images are silently undated. pub fn tiff_metadata(tiff_data: &[u8]) -> Result { let reader = TiffReader::new(tiff_data) .ok_or_else(|| crate::DecodeError::Metadata("malformed EXIF header".into()))?; let mut md = crate::Metadata::default(); // Fallback dates accumulate across *all* IFDs before being resolved. They // must not be settled per-IFD: one reference scanner writes `DateTime` in // the main IFD and `DateTimeDigitized` in the Exif sub-IFD, 102 seconds // apart, so resolving after the first would take the worse of the two. let mut fb = FallbackDates::default(); for ifd in reader.ifd_offsets() { let Some(entries) = reader.read_ifd(ifd) else { continue; }; read_exif_entries(&reader, &entries, &mut md, &mut fb); // The interesting tags live in the Exif sub-IFD, which the main IFD // points at rather than containing. if let Some(e) = entries.iter().find(|e| e.tag == EXIF_IFD_POINTER) { if let Some(sub) = reader.scalar(e).and_then(|o| reader.read_ifd(o)) { read_exif_entries(&reader, &sub, &mut md, &mut fb); } } } // Ranked: when the shutter fired, else when the image was digitised, else // when the file was last written. `DateTime` moves on every re-save, so it // is the last resort rather than the first match. if md.captured_at.is_none() { md.captured_at = fb.digitized.or(fb.modified); } Ok(md) } /// Offset of the TIFF header inside a JPEG's `APP1` EXIF segment. /// /// Walks the marker chain rather than scanning for the `Exif\0\0` magic: /// scanning could match those bytes inside compressed image data and point the /// TIFF reader at noise. fn find_exif_tiff(bytes: &[u8]) -> Option { if !bytes.starts_with(&[0xFF, 0xD8]) { return None; } let mut i = 2; // Bounded by the header slice callers pass; a malformed length field // cannot walk past the end because every read is checked. while i + 4 <= bytes.len() { if bytes[i] != 0xFF { return None; } let marker = bytes[i + 1]; // Start of scan: image data follows, and no more headers. if marker == 0xDA { return None; } let len = u16::from_be_bytes([bytes[i + 2], bytes[i + 3]]) as usize; if len < 2 { return None; } // APP1 carrying the "Exif\0\0" identifier. if marker == 0xE1 { let seg = bytes.get(i + 4..i + 2 + len)?; if seg.starts_with(b"Exif\0\0") { return Some(i + 4 + 6); } } i += 2 + len; } None } /// Exif sub-IFD pointer, where the capture tags actually live. const EXIF_IFD_POINTER: u16 = 0x8769; mod exif_tag { pub const MAKE: u16 = 0x010F; pub const MODEL: u16 = 0x0110; /// How the stored pixels sit relative to how the image should be seen. /// /// Lives in the main IFD rather than the Exif sub-IFD, which is why it is /// found at all: the sub-IFD is where the *capture* tags are. pub const ORIENTATION: u16 = 0x0112; /// When the shutter fired. Absent on scanner output. pub const DATE_TIME_ORIGINAL: u16 = 0x9003; /// When the file was written. A camera sets both; a **scanner sets only /// this one**, so without it every scanned frame is undated — 5,712 of /// them in this project's reference library. pub const DATE_TIME: u16 = 0x0132; /// Digitisation time. Another fallback some devices fill instead. pub const DATE_TIME_DIGITIZED: u16 = 0x9004; pub const OFFSET_TIME_ORIGINAL: u16 = 0x9011; pub const ISO: u16 = 0x8827; pub const LENS_MODEL: u16 = 0xA434; pub const PIXEL_X: u16 = 0xA002; pub const PIXEL_Y: u16 = 0xA003; } /// Dates that stand in for a missing `DateTimeOriginal`. /// /// Collected across every IFD and ranked once at the end, because the two can /// live in different IFDs and disagree. #[derive(Default)] struct FallbackDates { /// When the image was digitised. A scanner's real capture time. digitized: Option, /// When the file was last written. Moves on re-save, so lowest rank. modified: Option, } fn read_exif_entries( r: &TiffReader, entries: &[Entry], md: &mut crate::Metadata, fb: &mut FallbackDates, ) { for e in entries { match e.tag { exif_tag::MAKE => md.make = r.ascii(e), exif_tag::MODEL => md.model = r.ascii(e), exif_tag::LENS_MODEL => md.lens = r.ascii(e), exif_tag::ISO => md.iso = r.scalar(e), exif_tag::PIXEL_X => md.width = r.scalar(e), exif_tag::PIXEL_Y => md.height = r.scalar(e), // First IFD wins, unlike the fields above, which take the last // reading. This loop visits every IFD in the file, and a TIFF's // second one describes the *embedded thumbnail* — which some // bodies write already upright, tagged `1`. Letting that overwrite // the main image's tag would lay every portrait frame on its side. exif_tag::ORIENTATION => { if let Some(v) = r.scalar(e) { md.orientation .get_or_insert_with(|| dr_types::Orientation::from_exif(v as u16)); } } exif_tag::DATE_TIME_ORIGINAL => { if let Some(t) = r.ascii(e).as_deref().and_then(crate::parse_exif_datetime) { md.captured_at = Some(t); } } exif_tag::DATE_TIME_DIGITIZED => { if let Some(t) = r.ascii(e).as_deref().and_then(crate::parse_exif_datetime) { fb.digitized.get_or_insert(t); } } exif_tag::DATE_TIME => { if let Some(t) = r.ascii(e).as_deref().and_then(crate::parse_exif_datetime) { fb.modified.get_or_insert(t); } } exif_tag::OFFSET_TIME_ORIGINAL => { md.captured_offset = r.ascii(e).as_deref().and_then(crate::parse_exif_offset) } _ => {} } } } /// Whether a byte slice is a complete JPEG. /// /// A truncated JPEG decodes to a partial image rather than an error — the /// exact failure that made range-fetched thumbnails render as the top tenth of /// the frame. Checking for the end-of-image marker catches it before the /// result reaches a cache or a screen. pub fn is_complete_jpeg(bytes: &[u8]) -> bool { bytes.len() > 4 && bytes.starts_with(&[0xFF, 0xD8]) // Trailing padding after EOI is legal and does occur, so scan the tail // rather than testing only the final two bytes. && bytes .rchunks(64) .next() .map(|tail| tail.windows(2).any(|w| w == [0xFF, 0xD9])) .unwrap_or(false) } #[cfg(test)] mod tests { use super::*; /// Build a little-endian TIFF header with one IFD. fn tiff(entries: &[(u16, u16, u32, u32)], next_ifd: u32) -> Vec { let mut v = Vec::new(); v.extend_from_slice(b"II"); v.extend_from_slice(&42u16.to_le_bytes()); v.extend_from_slice(&8u32.to_le_bytes()); // first IFD at offset 8 v.extend_from_slice(&(entries.len() as u16).to_le_bytes()); for (tag, kind, count, value) in entries { v.extend_from_slice(&tag.to_le_bytes()); v.extend_from_slice(&kind.to_le_bytes()); v.extend_from_slice(&count.to_le_bytes()); v.extend_from_slice(&value.to_le_bytes()); } v.extend_from_slice(&next_ifd.to_le_bytes()); v.resize(v.len().max(1024), 0); v } /// Build a little-endian TIFF with two chained IFDs. /// /// The second one stands in for a TIFF's thumbnail IFD, which is where the /// orientation test's whole point lives. fn tiff_two_ifds(first: &[(u16, u16, u32, u32)], second: &[(u16, u16, u32, u32)]) -> Vec { // IFD0 occupies 2 + 12n + 4 bytes from offset 8. let second_at = 8 + 2 + 12 * first.len() as u32 + 4; let mut v = tiff(first, second_at); v.truncate(second_at as usize); v.extend_from_slice(&(second.len() as u16).to_le_bytes()); for (tag, kind, count, value) in second { v.extend_from_slice(&tag.to_le_bytes()); v.extend_from_slice(&kind.to_le_bytes()); v.extend_from_slice(&count.to_le_bytes()); v.extend_from_slice(&value.to_le_bytes()); } v.extend_from_slice(&0u32.to_le_bytes()); v.resize(v.len().max(1024), 0); v } #[test] fn the_grid_reads_a_jpegs_orientation_without_decoding_it() { // The exact call the thumbnail worker makes, on the exact bytes it // has: a header, no pixels. Going through `metadata` instead would // build a rawler decoder per grid cell. let jpeg = jpeg_with_exif(&[(exif_tag::ORIENTATION, 3, 1, 6)], &[]); assert_eq!( crate::orientation(&jpeg), Some(dr_types::Orientation::from_exif(6)) ); // And a file that says nothing declines rather than guessing. let plain = jpeg_with_exif(&[(exif_tag::ISO, 3, 1, 400)], &[]); assert_eq!(crate::orientation(&plain), None); } #[test] fn orientation_is_read_from_the_main_ifd() { // 6 is "rotate 90° clockwise to display" — a phone or a body held on // its side, which is the case this whole path exists for. let h = tiff(&[(exif_tag::ORIENTATION, 3, 1, 6)], 0); let md = tiff_metadata(&h).expect("metadata"); assert_eq!(md.orientation, Some(dr_types::Orientation::from_exif(6))); } #[test] fn a_file_with_no_orientation_tag_reports_none_rather_than_upright() { // "Nothing was said" and "the camera was level" are different claims. // They are displayed alike, but only one of them can later be // distinguished from a deliberate `1`. let h = tiff(&[(tag::IMAGE_WIDTH, 4, 1, 1620)], 0); let md = tiff_metadata(&h).expect("metadata"); assert_eq!(md.orientation, None); } #[test] fn the_thumbnail_ifd_does_not_overwrite_the_main_images_orientation() { // The regression this guards: some bodies write their embedded // thumbnail already upright and tag that IFD `1`. Reading every IFD // last-wins — which is right for make, model and the dates — would // take the thumbnail's `1` and lay every portrait frame on its side. let h = tiff_two_ifds( &[(exif_tag::ORIENTATION, 3, 1, 8)], &[(exif_tag::ORIENTATION, 3, 1, 1)], ); let md = tiff_metadata(&h).expect("metadata"); assert_eq!(md.orientation, Some(dr_types::Orientation::from_exif(8))); } #[test] fn finds_a_jpeg_interchange_preview() { let h = tiff( &[ (tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 100_000), (tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 1_500_000), (tag::IMAGE_WIDTH, 4, 1, 1620), (tag::IMAGE_LENGTH, 4, 1, 1080), ], 0, ); let loc = locate_preview(&h, 25_000_000).expect("a preview"); assert_eq!(loc.range, 100_000..1_600_000); assert_eq!(loc.width, Some(1620)); assert_eq!(loc.height, Some(1080)); } #[test] fn a_range_past_the_end_of_file_is_rejected() { // The check that stops a corrupt offset becoming a wild range request. let h = tiff( &[ (tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 20_000_000), (tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 10_000_000), ], 0, ); assert!(locate_preview(&h, 25_000_000).is_none()); } #[test] fn a_preview_larger_than_half_the_file_is_rejected() { // That is the full image mislabelled; "locating" it would transfer the // whole file, which is what this exists to avoid. let h = tiff( &[ (tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 100), (tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 9_000_000), ], 0, ); assert!(locate_preview(&h, 10_000_000).is_none()); } #[test] fn a_zero_length_preview_is_rejected() { let h = tiff( &[ (tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 100), (tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 0), ], 0, ); assert!(locate_preview(&h, 1_000_000).is_none()); } #[test] fn uncompressed_strips_are_not_mistaken_for_a_preview() { // Raw sensor data lives in strips too; feeding it to a JPEG decoder // would produce noise. let h = tiff( &[ (tag::STRIP_OFFSETS, 4, 1, 50_000), (tag::STRIP_BYTE_COUNTS, 4, 1, 800_000), (tag::COMPRESSION, 3, 1, 1), // uncompressed ], 0, ); assert!(locate_preview(&h, 25_000_000).is_none()); } #[test] fn jpeg_compressed_strips_are_accepted() { let h = tiff( &[ (tag::STRIP_OFFSETS, 4, 1, 50_000), (tag::STRIP_BYTE_COUNTS, 4, 1, 800_000), (tag::COMPRESSION, 3, 1, COMPRESSION_JPEG), ], 0, ); let loc = locate_preview(&h, 25_000_000).expect("a preview"); assert_eq!(loc.range, 50_000..850_000); } #[test] fn multi_strip_images_are_rejected() { // Several strips means tiled sensor data, not one contiguous JPEG. let h = tiff( &[ (tag::STRIP_OFFSETS, 4, 8, 50_000), (tag::STRIP_BYTE_COUNTS, 4, 8, 800_000), (tag::COMPRESSION, 3, 1, COMPRESSION_JPEG), ], 0, ); assert!(locate_preview(&h, 25_000_000).is_none()); } #[test] fn big_endian_files_parse() { // Nikon and Olympus ship big-endian containers. let mut v = Vec::new(); v.extend_from_slice(b"MM"); v.extend_from_slice(&42u16.to_be_bytes()); v.extend_from_slice(&8u32.to_be_bytes()); v.extend_from_slice(&2u16.to_be_bytes()); for (tag, kind, count, value) in [ (tag::JPEG_INTERCHANGE_FORMAT, 4u16, 1u32, 4096u32), (tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 900_000), ] { v.extend_from_slice(&tag.to_be_bytes()); v.extend_from_slice(&kind.to_be_bytes()); v.extend_from_slice(&count.to_be_bytes()); v.extend_from_slice(&value.to_be_bytes()); } v.extend_from_slice(&0u32.to_be_bytes()); v.resize(1024, 0); let loc = locate_preview(&v, 25_000_000).expect("a preview"); assert_eq!(loc.range, 4096..904_096); } #[test] fn a_big_endian_short_reads_from_the_high_half() { // The classic TIFF trap: a SHORT is left-justified in the 4-byte value // field on big-endian, so reading it as a LONG yields a huge number. let mut v = Vec::new(); v.extend_from_slice(b"MM"); v.extend_from_slice(&42u16.to_be_bytes()); v.extend_from_slice(&8u32.to_be_bytes()); v.extend_from_slice(&4u16.to_be_bytes()); for (tag, kind, count, value) in [ (tag::STRIP_OFFSETS, 4u16, 1u32, 1000u32), (tag::STRIP_BYTE_COUNTS, 4, 1, 500_000), (tag::COMPRESSION, 3, 1, (COMPRESSION_JPEG) << 16), (tag::IMAGE_WIDTH, 3, 1, 1620u32 << 16), ] { v.extend_from_slice(&tag.to_be_bytes()); v.extend_from_slice(&kind.to_be_bytes()); v.extend_from_slice(&count.to_be_bytes()); v.extend_from_slice(&value.to_be_bytes()); } v.extend_from_slice(&0u32.to_be_bytes()); v.resize(1024, 0); let loc = locate_preview(&v, 25_000_000).expect("a preview"); assert_eq!(loc.width, Some(1620), "short read from the wrong half"); } #[test] fn the_largest_preview_wins_across_ifds() { // Cameras carry both a tiny thumbnail and a screen-sized preview; the // larger downscales better. let mut v = tiff( &[ (tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 1000), (tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 8_000), ], 200, ); // A second IFD at offset 200 with a much larger preview. let second = 200usize; v[second..second + 2].copy_from_slice(&2u16.to_le_bytes()); for (i, (tag, kind, count, value)) in [ (tag::JPEG_INTERCHANGE_FORMAT, 4u16, 1u32, 20_000u32), (tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 1_200_000), ] .iter() .enumerate() { let e = second + 2 + i * 12; v[e..e + 2].copy_from_slice(&tag.to_le_bytes()); v[e + 2..e + 4].copy_from_slice(&kind.to_le_bytes()); v[e + 4..e + 8].copy_from_slice(&count.to_le_bytes()); v[e + 8..e + 12].copy_from_slice(&value.to_le_bytes()); } let loc = locate_preview(&v, 25_000_000).expect("a preview"); assert_eq!(loc.len(), 1_200_000, "should pick the larger"); } #[test] fn a_self_referential_ifd_chain_terminates() { // A malformed file pointing an IFD at itself must not hang the app. let h = tiff(&[(tag::IMAGE_WIDTH, 4, 1, 100)], 8); let _ = locate_preview(&h, 1_000_000); } #[test] fn an_absurd_entry_count_is_rejected_not_allocated() { let mut v = Vec::new(); v.extend_from_slice(b"II"); v.extend_from_slice(&42u16.to_le_bytes()); v.extend_from_slice(&8u32.to_le_bytes()); v.extend_from_slice(&60000u16.to_le_bytes()); // claims 60k entries v.resize(1024, 0); assert!(locate_preview(&v, 1_000_000).is_none()); } #[test] fn non_tiff_input_declines_cleanly() { assert!(locate_preview(b"not a tiff at all", 1000).is_none()); assert!(locate_preview(&[], 1000).is_none()); // CR3 is ISO-BMFF, not TIFF — declining is correct. assert!(locate_preview(b"\0\0\0\x18ftypcrx ", 1000).is_none()); } /// Build a JPEG carrying an APP1 EXIF block with the given IFD entries. fn jpeg_with_exif(entries: &[(u16, u16, u32, u32)], extra: &[u8]) -> Vec { let mut tiff = Vec::new(); tiff.extend_from_slice(b"II"); tiff.extend_from_slice(&42u16.to_le_bytes()); tiff.extend_from_slice(&8u32.to_le_bytes()); tiff.extend_from_slice(&(entries.len() as u16).to_le_bytes()); for (tag, kind, count, value) in entries { tiff.extend_from_slice(&tag.to_le_bytes()); tiff.extend_from_slice(&kind.to_le_bytes()); tiff.extend_from_slice(&count.to_le_bytes()); tiff.extend_from_slice(&value.to_le_bytes()); } tiff.extend_from_slice(&0u32.to_le_bytes()); tiff.extend_from_slice(extra); let payload_len = (tiff.len() + 6 + 2) as u16; let mut out = vec![0xFF, 0xD8, 0xFF, 0xE1]; out.extend_from_slice(&payload_len.to_be_bytes()); out.extend_from_slice(b"Exif\0\0"); out.extend_from_slice(&tiff); out } #[test] fn jpeg_exif_yields_a_capture_time() { // rawler decodes no JPEG at all, so without this path every JPEG in a // library is undated — 5,712 scanned frames in the reference library. let date = b"2013:06:28 23:32:54\0"; let mut extra = Vec::new(); let date_offset = 8 + 2 + 12 + 4; extra.extend_from_slice(date); let jpeg = jpeg_with_exif( &[( exif_tag::DATE_TIME_ORIGINAL, 2, date.len() as u32, date_offset as u32, )], &extra, ); let md = jpeg_metadata(&jpeg).expect("EXIF"); assert_eq!(md.captured_at, Some(1_372_462_374)); } #[test] fn a_jpeg_without_exif_reports_no_segment() { assert!(jpeg_metadata(&[0xFF, 0xD8, 0xFF, 0xDA, 0, 2]).is_err()); assert!(jpeg_metadata(b"not a jpeg").is_err()); } #[test] fn ascii_values_lose_their_nul_padding() { // A model name with a trailing NUL compares unequal to the same name // without one, which would split one camera into two in any grouping. let model = b"CanoScan 9000F Mark II\0"; let mut extra = Vec::new(); let off = 8 + 2 + 12 + 4; extra.extend_from_slice(model); let jpeg = jpeg_with_exif( &[(exif_tag::MODEL, 2, model.len() as u32, off as u32)], &extra, ); let md = jpeg_metadata(&jpeg).expect("EXIF"); assert_eq!(md.model.as_deref(), Some("CanoScan 9000F Mark II")); } #[test] fn a_marker_walk_does_not_run_off_a_truncated_file() { // Untrusted input (NFR-SEC-1): a length field claiming more than the // file holds must not read past the end. let mut jpeg = vec![0xFF, 0xD8, 0xFF, 0xE1]; jpeg.extend_from_slice(&60000u16.to_be_bytes()); jpeg.extend_from_slice(b"Exif\0\0"); assert!(jpeg_metadata(&jpeg).is_err()); } #[test] fn truncated_jpegs_are_detected() { // The bug this whole module exists to fix: a short read decodes to a // partial image rather than failing, so it must be caught by // inspection. let mut complete = vec![0xFF, 0xD8]; complete.extend_from_slice(&[0x00; 200]); complete.extend_from_slice(&[0xFF, 0xD9]); assert!(is_complete_jpeg(&complete)); let truncated = &complete[..complete.len() - 2]; assert!(!is_complete_jpeg(truncated)); } #[test] fn non_jpeg_bytes_are_not_complete_jpegs() { assert!(!is_complete_jpeg(&[])); assert!(!is_complete_jpeg(&[0xFF, 0xD9])); assert!(!is_complete_jpeg(b"PNG\r\n")); } }