Locate embedded previews by parsing the container header
Remote browsing must not transfer whole RAW files. The obvious shortcut — fetch a fixed prefix and hope the preview is inside — does not work: an embedded JPEG typically starts a few hundred KB in and runs for one to three MB, so a truncated fetch yields a JPEG whose scanlines stop partway down. Decoders render what they have rather than erroring, so the failure looks like a corrupt image rather than a short read. This reads the TIFF-structured containers — CR2, NEF, ARW, DNG, ORF — and returns an exact byte range for a Range: request. CR3 is ISO-BMFF and declines to the caller's whole-file path, which is correct if slower; a locator returning a wrong range would be far worse than one that declines. Every offset read from the file is bounds-checked against the real file length rather than trusted (NFR-SEC-1). Assisted-by: LLM
This commit is contained in:
@@ -0,0 +1,886 @@
|
||||
//! TRACES: FR-NC-3 | FR-CULL-2
|
||||
//! Finding an embedded preview's byte range from a file header alone.
|
||||
//!
|
||||
//! # Why this exists
|
||||
//!
|
||||
//! Remote browsing must not transfer whole RAW files (FR-NC-3). The obvious
|
||||
//! shortcut — fetch a fixed prefix and hope the preview is inside it — does
|
||||
//! not work: an embedded JPEG typically starts a few hundred KB in and runs
|
||||
//! for one to three MB, so a truncated fetch yields a JPEG whose scanlines
|
||||
//! stop partway down. Decoders render what they have rather than erroring, so
|
||||
//! the failure looks like a corrupt image, not a short read.
|
||||
//!
|
||||
//! What FR-NC-3 actually specifies is two-stage: read the header, *parse the
|
||||
//! container* to locate the preview, then fetch exactly those bytes.
|
||||
//!
|
||||
//! # Scope
|
||||
//!
|
||||
//! This reads TIFF-structured containers — CR2, NEF, ARW, DNG, and ORF all
|
||||
//! carry their previews in IFD entries. CR3 is ISO-BMFF and is not handled
|
||||
//! here; it falls back to the caller's whole-file path, which is correct if
|
||||
//! slower. A locator that returned a wrong range would be far worse than one
|
||||
//! that declines.
|
||||
//!
|
||||
//! Every offset read from the file is treated as hostile (NFR-SEC-1): bounds
|
||||
//! are checked against the real file length, never trusted.
|
||||
|
||||
use std::ops::Range;
|
||||
|
||||
/// Where a preview lives inside its container.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct PreviewLocation {
|
||||
/// Byte range of the JPEG, ready to hand to a `Range:` request.
|
||||
pub range: Range<u64>,
|
||||
/// Pixel dimensions where the container declared them. Used to pick the
|
||||
/// largest preview that is still smaller than a full decode.
|
||||
pub width: Option<u32>,
|
||||
pub height: Option<u32>,
|
||||
}
|
||||
|
||||
impl PreviewLocation {
|
||||
pub fn len(&self) -> u64 {
|
||||
self.range.end - self.range.start
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.range.end <= self.range.start
|
||||
}
|
||||
}
|
||||
|
||||
/// How many bytes of header a caller should fetch before calling this.
|
||||
///
|
||||
/// Large enough to cover the IFD chain in the formats measured, small enough
|
||||
/// that a miss costs little. Metadata alone needs less, but IFD1/IFD2 entries
|
||||
/// for the preview sit further in on some bodies.
|
||||
pub const HEADER_BYTES: u64 = 256 * 1024;
|
||||
|
||||
/// Locate the largest embedded preview at or below `max_edge`, if any.
|
||||
///
|
||||
/// `header` is the first [`HEADER_BYTES`] of the file; `file_len` is the whole
|
||||
/// file's length, needed to reject offsets that point past the end.
|
||||
///
|
||||
/// Returns `None` when the container is not TIFF-structured, declares no
|
||||
/// preview, or declares one whose range is not credible.
|
||||
pub fn locate_preview(header: &[u8], file_len: u64) -> Option<PreviewLocation> {
|
||||
let tiff = TiffReader::new(header)?;
|
||||
let mut best: Option<PreviewLocation> = None;
|
||||
|
||||
for ifd_offset in tiff.ifd_offsets() {
|
||||
let Some(entries) = tiff.read_ifd(ifd_offset) else {
|
||||
continue;
|
||||
};
|
||||
|
||||
if let Some(loc) = preview_from_entries(&tiff, &entries, file_len) {
|
||||
// Prefer the largest, since a bigger preview downscales better —
|
||||
// but anything is better than nothing.
|
||||
let better = match (&best, &loc) {
|
||||
(None, _) => true,
|
||||
(Some(b), l) => l.len() > b.len(),
|
||||
};
|
||||
if better {
|
||||
best = Some(loc);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
best
|
||||
}
|
||||
|
||||
/// TIFF tags that carry a preview's location.
|
||||
mod tag {
|
||||
/// Legacy thumbnail offset/length (IFD1 in most makes).
|
||||
pub const JPEG_INTERCHANGE_FORMAT: u16 = 0x0201;
|
||||
pub const JPEG_INTERCHANGE_FORMAT_LENGTH: u16 = 0x0202;
|
||||
/// Strip-based storage, which is how DNG and some NEF previews are held.
|
||||
pub const STRIP_OFFSETS: u16 = 0x0111;
|
||||
pub const STRIP_BYTE_COUNTS: u16 = 0x0117;
|
||||
pub const IMAGE_WIDTH: u16 = 0x0100;
|
||||
pub const IMAGE_LENGTH: u16 = 0x0101;
|
||||
/// 1 = full-resolution sensor data, 0 = a reduced-resolution preview.
|
||||
pub const NEW_SUBFILE_TYPE: u16 = 0x00FE;
|
||||
pub const COMPRESSION: u16 = 0x0103;
|
||||
pub const SUB_IFDS: u16 = 0x014A;
|
||||
}
|
||||
|
||||
/// JPEG compression, as opposed to raw sensor data.
|
||||
const COMPRESSION_JPEG: u32 = 6;
|
||||
const COMPRESSION_OLD_JPEG: u32 = 7;
|
||||
|
||||
fn preview_from_entries(
|
||||
tiff: &TiffReader,
|
||||
entries: &[Entry],
|
||||
file_len: u64,
|
||||
) -> Option<PreviewLocation> {
|
||||
let get = |t: u16| entries.iter().find(|e| e.tag == t);
|
||||
|
||||
// Reject the full-resolution image: it is sensor data, not a preview, and
|
||||
// "locating" it would transfer the whole file — the exact cost this avoids.
|
||||
if let Some(e) = get(tag::NEW_SUBFILE_TYPE) {
|
||||
if tiff.scalar(e)? == 0 && get(tag::JPEG_INTERCHANGE_FORMAT).is_none() {
|
||||
// Subfile type 0 means full resolution. Only continue if it is a
|
||||
// JPEG interchange entry, which a main image never is.
|
||||
return None;
|
||||
}
|
||||
}
|
||||
|
||||
// Strip-based entries must be JPEG-compressed; an uncompressed strip is
|
||||
// raw sensor data that no JPEG decoder will read.
|
||||
let (offset, length) = if let (Some(o), Some(l)) = (
|
||||
get(tag::JPEG_INTERCHANGE_FORMAT),
|
||||
get(tag::JPEG_INTERCHANGE_FORMAT_LENGTH),
|
||||
) {
|
||||
(tiff.scalar(o)? as u64, tiff.scalar(l)? as u64)
|
||||
} else if let (Some(o), Some(l), Some(c)) = (
|
||||
get(tag::STRIP_OFFSETS),
|
||||
get(tag::STRIP_BYTE_COUNTS),
|
||||
get(tag::COMPRESSION),
|
||||
) {
|
||||
let compression = tiff.scalar(c)?;
|
||||
if compression != COMPRESSION_JPEG && compression != COMPRESSION_OLD_JPEG {
|
||||
return None;
|
||||
}
|
||||
// A multi-strip image is tiled sensor data, not a single JPEG.
|
||||
if o.count != 1 || l.count != 1 {
|
||||
return None;
|
||||
}
|
||||
(tiff.scalar(o)? as u64, tiff.scalar(l)? as u64)
|
||||
} else {
|
||||
return None;
|
||||
};
|
||||
|
||||
// Everything below is validation against a hostile file (NFR-SEC-1).
|
||||
if length == 0 {
|
||||
return None;
|
||||
}
|
||||
let end = offset.checked_add(length)?;
|
||||
if end > file_len {
|
||||
return None;
|
||||
}
|
||||
// A "preview" the size of the whole file is the full image mislabelled.
|
||||
if length > file_len / 2 {
|
||||
return None;
|
||||
}
|
||||
|
||||
Some(PreviewLocation {
|
||||
range: offset..end,
|
||||
width: get(tag::IMAGE_WIDTH).and_then(|e| tiff.scalar(e)),
|
||||
height: get(tag::IMAGE_LENGTH).and_then(|e| tiff.scalar(e)),
|
||||
})
|
||||
}
|
||||
|
||||
/// One IFD entry.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
struct Entry {
|
||||
tag: u16,
|
||||
kind: u16,
|
||||
count: u32,
|
||||
/// The raw 4-byte value field — either the value itself or an offset to it.
|
||||
value: u32,
|
||||
}
|
||||
|
||||
/// A minimal TIFF structure reader.
|
||||
///
|
||||
/// Deliberately not a general TIFF parser: it reads the IFD chain and entry
|
||||
/// values and nothing else, because that is all locating a preview needs.
|
||||
struct TiffReader<'a> {
|
||||
data: &'a [u8],
|
||||
little_endian: bool,
|
||||
first_ifd: u32,
|
||||
}
|
||||
|
||||
impl<'a> TiffReader<'a> {
|
||||
fn new(data: &'a [u8]) -> Option<Self> {
|
||||
if data.len() < 8 {
|
||||
return None;
|
||||
}
|
||||
let little_endian = match &data[0..2] {
|
||||
b"II" => true,
|
||||
b"MM" => false,
|
||||
_ => return None,
|
||||
};
|
||||
|
||||
let magic = read_u16(data, 2, little_endian)?;
|
||||
// 42 is TIFF; 0x4F52 and 0x5352 are ORF's variants, which are
|
||||
// otherwise TIFF-shaped.
|
||||
if magic != 42 && magic != 0x4F52 && magic != 0x5352 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let first_ifd = read_u32(data, 4, little_endian)?;
|
||||
Some(Self {
|
||||
data,
|
||||
little_endian,
|
||||
first_ifd,
|
||||
})
|
||||
}
|
||||
|
||||
/// Every IFD worth searching: the chain from the header, plus any SubIFDs.
|
||||
///
|
||||
/// Bounded, because a malformed file can point an IFD at itself and a
|
||||
/// naive walk would never terminate.
|
||||
fn ifd_offsets(&self) -> Vec<u32> {
|
||||
const MAX_IFDS: usize = 16;
|
||||
let mut out = Vec::new();
|
||||
let mut seen = std::collections::HashSet::new();
|
||||
let mut next = self.first_ifd;
|
||||
|
||||
while next != 0 && out.len() < MAX_IFDS && seen.insert(next) {
|
||||
out.push(next);
|
||||
|
||||
// SubIFDs hold the preview in DNG and several NEF variants.
|
||||
if let Some(entries) = self.read_ifd(next) {
|
||||
if let Some(sub) = entries.iter().find(|e| e.tag == tag::SUB_IFDS) {
|
||||
for offset in self.offsets(sub) {
|
||||
if out.len() < MAX_IFDS && seen.insert(offset) {
|
||||
out.push(offset);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
match self.next_ifd_offset(next) {
|
||||
Some(n) => next = n,
|
||||
None => break,
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
fn read_ifd(&self, offset: u32) -> Option<Vec<Entry>> {
|
||||
let base = offset as usize;
|
||||
let count = read_u16(self.data, base, self.little_endian)? as usize;
|
||||
|
||||
// A plausible IFD has tens of entries, not thousands. A huge count is
|
||||
// a corrupt or hostile file, and allocating for it is the bug.
|
||||
if count > 512 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut entries = Vec::with_capacity(count);
|
||||
for i in 0..count {
|
||||
let e = base + 2 + i * 12;
|
||||
entries.push(Entry {
|
||||
tag: read_u16(self.data, e, self.little_endian)?,
|
||||
kind: read_u16(self.data, e + 2, self.little_endian)?,
|
||||
count: read_u32(self.data, e + 4, self.little_endian)?,
|
||||
value: read_u32(self.data, e + 8, self.little_endian)?,
|
||||
});
|
||||
}
|
||||
Some(entries)
|
||||
}
|
||||
|
||||
fn next_ifd_offset(&self, ifd: u32) -> Option<u32> {
|
||||
let base = ifd as usize;
|
||||
let count = read_u16(self.data, base, self.little_endian)? as usize;
|
||||
read_u32(self.data, base + 2 + count * 12, self.little_endian)
|
||||
}
|
||||
|
||||
/// An entry's value as a single number.
|
||||
///
|
||||
/// Handles the inline case only for the scalar types a preview entry uses;
|
||||
/// anything larger than four bytes is stored out of line and read through
|
||||
/// its offset.
|
||||
fn scalar(&self, e: &Entry) -> Option<u32> {
|
||||
match e.kind {
|
||||
// SHORT, inline when count is 1.
|
||||
3 if e.count == 1 => Some(if self.little_endian {
|
||||
e.value & 0xFFFF
|
||||
} else {
|
||||
// Big-endian packs a short into the high half of the field.
|
||||
e.value >> 16
|
||||
}),
|
||||
// LONG, always inline at count 1.
|
||||
4 if e.count == 1 => Some(e.value),
|
||||
// A count above one points elsewhere; take the first element.
|
||||
3 => read_u16(self.data, e.value as usize, self.little_endian).map(u32::from),
|
||||
4 => read_u32(self.data, e.value as usize, self.little_endian),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// An ASCII entry's string value.
|
||||
///
|
||||
/// EXIF strings are NUL-terminated and often padded, and camera vendors
|
||||
/// pad with spaces too — both are trimmed, since a model name with a
|
||||
/// trailing NUL compares unequal to the same name without one.
|
||||
fn ascii(&self, e: &Entry) -> Option<String> {
|
||||
// Type 2 is ASCII. Up to four bytes live inline; longer strings are
|
||||
// stored at the offset in the value field.
|
||||
if e.kind != 2 || e.count == 0 {
|
||||
return None;
|
||||
}
|
||||
let len = e.count as usize;
|
||||
let bytes = if len <= 4 {
|
||||
let raw = if self.little_endian {
|
||||
e.value.to_le_bytes()
|
||||
} else {
|
||||
e.value.to_be_bytes()
|
||||
};
|
||||
raw[..len.min(4)].to_vec()
|
||||
} else {
|
||||
self.data.get(e.value as usize..e.value as usize + len)?.to_vec()
|
||||
};
|
||||
|
||||
let s = String::from_utf8_lossy(&bytes);
|
||||
let s = s.trim_end_matches('\0').trim();
|
||||
if s.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(s.to_string())
|
||||
}
|
||||
}
|
||||
|
||||
/// An entry's values as a list of offsets (for SubIFDs).
|
||||
fn offsets(&self, e: &Entry) -> Vec<u32> {
|
||||
if e.kind != 4 {
|
||||
return Vec::new();
|
||||
}
|
||||
if e.count == 1 {
|
||||
return vec![e.value];
|
||||
}
|
||||
// Bounded: a SubIFD list is a handful of entries, never thousands.
|
||||
(0..e.count.min(8))
|
||||
.filter_map(|i| {
|
||||
read_u32(
|
||||
self.data,
|
||||
e.value as usize + (i as usize) * 4,
|
||||
self.little_endian,
|
||||
)
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
fn read_u16(data: &[u8], at: usize, le: bool) -> Option<u16> {
|
||||
let b = data.get(at..at + 2)?;
|
||||
Some(if le {
|
||||
u16::from_le_bytes([b[0], b[1]])
|
||||
} else {
|
||||
u16::from_be_bytes([b[0], b[1]])
|
||||
})
|
||||
}
|
||||
|
||||
fn read_u32(data: &[u8], at: usize, le: bool) -> Option<u32> {
|
||||
let b = data.get(at..at + 4)?;
|
||||
Some(if le {
|
||||
u32::from_le_bytes([b[0], b[1], b[2], b[3]])
|
||||
} else {
|
||||
u32::from_be_bytes([b[0], b[1], b[2], b[3]])
|
||||
})
|
||||
}
|
||||
|
||||
/// Read EXIF from a JPEG's APP1 segment.
|
||||
///
|
||||
/// A JPEG's EXIF block is a complete TIFF structure embedded in an `APP1`
|
||||
/// marker, so the reader above does the work — only finding the block differs.
|
||||
///
|
||||
/// This exists because rawler decodes no JPEG at all, and a photo library is
|
||||
/// full of them: camera JPEGs, and in this project's reference library nearly
|
||||
/// six thousand scanned frames. Without it every one is undated and missing
|
||||
/// from the timeline.
|
||||
pub fn jpeg_metadata(bytes: &[u8]) -> Result<crate::Metadata, crate::DecodeError> {
|
||||
let tiff_start = find_exif_tiff(bytes)
|
||||
.ok_or_else(|| crate::DecodeError::Metadata("no EXIF segment".into()))?;
|
||||
tiff_metadata(&bytes[tiff_start..])
|
||||
}
|
||||
|
||||
/// Read EXIF from a bare TIFF structure.
|
||||
///
|
||||
/// Serves two callers: a JPEG's APP1 payload, and a TIFF-derived RAW whose
|
||||
/// primary decoder returned no date. The second case is real — rawler reports
|
||||
/// no `DateTimeOriginal` for some DNGs whose tag sits plainly at byte 826 —
|
||||
/// and without this fallback those images are silently undated.
|
||||
pub fn tiff_metadata(tiff_data: &[u8]) -> Result<crate::Metadata, crate::DecodeError> {
|
||||
let reader = TiffReader::new(tiff_data)
|
||||
.ok_or_else(|| crate::DecodeError::Metadata("malformed EXIF header".into()))?;
|
||||
|
||||
let mut md = crate::Metadata::default();
|
||||
// Fallback dates accumulate across *all* IFDs before being resolved. They
|
||||
// must not be settled per-IFD: one reference scanner writes `DateTime` in
|
||||
// the main IFD and `DateTimeDigitized` in the Exif sub-IFD, 102 seconds
|
||||
// apart, so resolving after the first would take the worse of the two.
|
||||
let mut fb = FallbackDates::default();
|
||||
|
||||
for ifd in reader.ifd_offsets() {
|
||||
let Some(entries) = reader.read_ifd(ifd) else {
|
||||
continue;
|
||||
};
|
||||
read_exif_entries(&reader, &entries, &mut md, &mut fb);
|
||||
|
||||
// The interesting tags live in the Exif sub-IFD, which the main IFD
|
||||
// points at rather than containing.
|
||||
if let Some(e) = entries.iter().find(|e| e.tag == EXIF_IFD_POINTER) {
|
||||
if let Some(sub) = reader.scalar(e).and_then(|o| reader.read_ifd(o)) {
|
||||
read_exif_entries(&reader, &sub, &mut md, &mut fb);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Ranked: when the shutter fired, else when the image was digitised, else
|
||||
// when the file was last written. `DateTime` moves on every re-save, so it
|
||||
// is the last resort rather than the first match.
|
||||
if md.captured_at.is_none() {
|
||||
md.captured_at = fb.digitized.or(fb.modified);
|
||||
}
|
||||
Ok(md)
|
||||
}
|
||||
|
||||
/// Offset of the TIFF header inside a JPEG's `APP1` EXIF segment.
|
||||
///
|
||||
/// Walks the marker chain rather than scanning for the `Exif\0\0` magic:
|
||||
/// scanning could match those bytes inside compressed image data and point the
|
||||
/// TIFF reader at noise.
|
||||
fn find_exif_tiff(bytes: &[u8]) -> Option<usize> {
|
||||
if !bytes.starts_with(&[0xFF, 0xD8]) {
|
||||
return None;
|
||||
}
|
||||
let mut i = 2;
|
||||
// Bounded by the header slice callers pass; a malformed length field
|
||||
// cannot walk past the end because every read is checked.
|
||||
while i + 4 <= bytes.len() {
|
||||
if bytes[i] != 0xFF {
|
||||
return None;
|
||||
}
|
||||
let marker = bytes[i + 1];
|
||||
// Start of scan: image data follows, and no more headers.
|
||||
if marker == 0xDA {
|
||||
return None;
|
||||
}
|
||||
let len = u16::from_be_bytes([bytes[i + 2], bytes[i + 3]]) as usize;
|
||||
if len < 2 {
|
||||
return None;
|
||||
}
|
||||
// APP1 carrying the "Exif\0\0" identifier.
|
||||
if marker == 0xE1 {
|
||||
let seg = bytes.get(i + 4..i + 2 + len)?;
|
||||
if seg.starts_with(b"Exif\0\0") {
|
||||
return Some(i + 4 + 6);
|
||||
}
|
||||
}
|
||||
i += 2 + len;
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
/// Exif sub-IFD pointer, where the capture tags actually live.
|
||||
const EXIF_IFD_POINTER: u16 = 0x8769;
|
||||
|
||||
mod exif_tag {
|
||||
pub const MAKE: u16 = 0x010F;
|
||||
pub const MODEL: u16 = 0x0110;
|
||||
/// When the shutter fired. Absent on scanner output.
|
||||
pub const DATE_TIME_ORIGINAL: u16 = 0x9003;
|
||||
/// When the file was written. A camera sets both; a **scanner sets only
|
||||
/// this one**, so without it every scanned frame is undated — 5,712 of
|
||||
/// them in this project's reference library.
|
||||
pub const DATE_TIME: u16 = 0x0132;
|
||||
/// Digitisation time. Another fallback some devices fill instead.
|
||||
pub const DATE_TIME_DIGITIZED: u16 = 0x9004;
|
||||
pub const OFFSET_TIME_ORIGINAL: u16 = 0x9011;
|
||||
pub const ISO: u16 = 0x8827;
|
||||
pub const LENS_MODEL: u16 = 0xA434;
|
||||
pub const PIXEL_X: u16 = 0xA002;
|
||||
pub const PIXEL_Y: u16 = 0xA003;
|
||||
}
|
||||
|
||||
/// Dates that stand in for a missing `DateTimeOriginal`.
|
||||
///
|
||||
/// Collected across every IFD and ranked once at the end, because the two can
|
||||
/// live in different IFDs and disagree.
|
||||
#[derive(Default)]
|
||||
struct FallbackDates {
|
||||
/// When the image was digitised. A scanner's real capture time.
|
||||
digitized: Option<i64>,
|
||||
/// When the file was last written. Moves on re-save, so lowest rank.
|
||||
modified: Option<i64>,
|
||||
}
|
||||
|
||||
fn read_exif_entries(
|
||||
r: &TiffReader,
|
||||
entries: &[Entry],
|
||||
md: &mut crate::Metadata,
|
||||
fb: &mut FallbackDates,
|
||||
) {
|
||||
|
||||
for e in entries {
|
||||
match e.tag {
|
||||
exif_tag::MAKE => md.make = r.ascii(e),
|
||||
exif_tag::MODEL => md.model = r.ascii(e),
|
||||
exif_tag::LENS_MODEL => md.lens = r.ascii(e),
|
||||
exif_tag::ISO => md.iso = r.scalar(e),
|
||||
exif_tag::PIXEL_X => md.width = r.scalar(e),
|
||||
exif_tag::PIXEL_Y => md.height = r.scalar(e),
|
||||
exif_tag::DATE_TIME_ORIGINAL => {
|
||||
if let Some(t) = r.ascii(e).as_deref().and_then(crate::parse_exif_datetime) {
|
||||
md.captured_at = Some(t);
|
||||
}
|
||||
}
|
||||
exif_tag::DATE_TIME_DIGITIZED => {
|
||||
if let Some(t) = r.ascii(e).as_deref().and_then(crate::parse_exif_datetime) {
|
||||
fb.digitized.get_or_insert(t);
|
||||
}
|
||||
}
|
||||
exif_tag::DATE_TIME => {
|
||||
if let Some(t) = r.ascii(e).as_deref().and_then(crate::parse_exif_datetime) {
|
||||
fb.modified.get_or_insert(t);
|
||||
}
|
||||
}
|
||||
exif_tag::OFFSET_TIME_ORIGINAL => {
|
||||
md.captured_offset = r.ascii(e).as_deref().and_then(crate::parse_exif_offset)
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether a byte slice is a complete JPEG.
|
||||
///
|
||||
/// A truncated JPEG decodes to a partial image rather than an error — the
|
||||
/// exact failure that made range-fetched thumbnails render as the top tenth of
|
||||
/// the frame. Checking for the end-of-image marker catches it before the
|
||||
/// result reaches a cache or a screen.
|
||||
pub fn is_complete_jpeg(bytes: &[u8]) -> bool {
|
||||
bytes.len() > 4
|
||||
&& bytes.starts_with(&[0xFF, 0xD8])
|
||||
// Trailing padding after EOI is legal and does occur, so scan the tail
|
||||
// rather than testing only the final two bytes.
|
||||
&& bytes
|
||||
.rchunks(64)
|
||||
.next()
|
||||
.map(|tail| tail.windows(2).any(|w| w == [0xFF, 0xD9]))
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// Build a little-endian TIFF header with one IFD.
|
||||
fn tiff(entries: &[(u16, u16, u32, u32)], next_ifd: u32) -> Vec<u8> {
|
||||
let mut v = Vec::new();
|
||||
v.extend_from_slice(b"II");
|
||||
v.extend_from_slice(&42u16.to_le_bytes());
|
||||
v.extend_from_slice(&8u32.to_le_bytes()); // first IFD at offset 8
|
||||
|
||||
v.extend_from_slice(&(entries.len() as u16).to_le_bytes());
|
||||
for (tag, kind, count, value) in entries {
|
||||
v.extend_from_slice(&tag.to_le_bytes());
|
||||
v.extend_from_slice(&kind.to_le_bytes());
|
||||
v.extend_from_slice(&count.to_le_bytes());
|
||||
v.extend_from_slice(&value.to_le_bytes());
|
||||
}
|
||||
v.extend_from_slice(&next_ifd.to_le_bytes());
|
||||
v.resize(v.len().max(1024), 0);
|
||||
v
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn finds_a_jpeg_interchange_preview() {
|
||||
let h = tiff(
|
||||
&[
|
||||
(tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 100_000),
|
||||
(tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 1_500_000),
|
||||
(tag::IMAGE_WIDTH, 4, 1, 1620),
|
||||
(tag::IMAGE_LENGTH, 4, 1, 1080),
|
||||
],
|
||||
0,
|
||||
);
|
||||
let loc = locate_preview(&h, 25_000_000).expect("a preview");
|
||||
assert_eq!(loc.range, 100_000..1_600_000);
|
||||
assert_eq!(loc.width, Some(1620));
|
||||
assert_eq!(loc.height, Some(1080));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_range_past_the_end_of_file_is_rejected() {
|
||||
// The check that stops a corrupt offset becoming a wild range request.
|
||||
let h = tiff(
|
||||
&[
|
||||
(tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 20_000_000),
|
||||
(tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 10_000_000),
|
||||
],
|
||||
0,
|
||||
);
|
||||
assert!(locate_preview(&h, 25_000_000).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_preview_larger_than_half_the_file_is_rejected() {
|
||||
// That is the full image mislabelled; "locating" it would transfer the
|
||||
// whole file, which is what this exists to avoid.
|
||||
let h = tiff(
|
||||
&[
|
||||
(tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 100),
|
||||
(tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 9_000_000),
|
||||
],
|
||||
0,
|
||||
);
|
||||
assert!(locate_preview(&h, 10_000_000).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_zero_length_preview_is_rejected() {
|
||||
let h = tiff(
|
||||
&[
|
||||
(tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 100),
|
||||
(tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 0),
|
||||
],
|
||||
0,
|
||||
);
|
||||
assert!(locate_preview(&h, 1_000_000).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn uncompressed_strips_are_not_mistaken_for_a_preview() {
|
||||
// Raw sensor data lives in strips too; feeding it to a JPEG decoder
|
||||
// would produce noise.
|
||||
let h = tiff(
|
||||
&[
|
||||
(tag::STRIP_OFFSETS, 4, 1, 50_000),
|
||||
(tag::STRIP_BYTE_COUNTS, 4, 1, 800_000),
|
||||
(tag::COMPRESSION, 3, 1, 1), // uncompressed
|
||||
],
|
||||
0,
|
||||
);
|
||||
assert!(locate_preview(&h, 25_000_000).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn jpeg_compressed_strips_are_accepted() {
|
||||
let h = tiff(
|
||||
&[
|
||||
(tag::STRIP_OFFSETS, 4, 1, 50_000),
|
||||
(tag::STRIP_BYTE_COUNTS, 4, 1, 800_000),
|
||||
(tag::COMPRESSION, 3, 1, COMPRESSION_JPEG),
|
||||
],
|
||||
0,
|
||||
);
|
||||
let loc = locate_preview(&h, 25_000_000).expect("a preview");
|
||||
assert_eq!(loc.range, 50_000..850_000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multi_strip_images_are_rejected() {
|
||||
// Several strips means tiled sensor data, not one contiguous JPEG.
|
||||
let h = tiff(
|
||||
&[
|
||||
(tag::STRIP_OFFSETS, 4, 8, 50_000),
|
||||
(tag::STRIP_BYTE_COUNTS, 4, 8, 800_000),
|
||||
(tag::COMPRESSION, 3, 1, COMPRESSION_JPEG),
|
||||
],
|
||||
0,
|
||||
);
|
||||
assert!(locate_preview(&h, 25_000_000).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn big_endian_files_parse() {
|
||||
// Nikon and Olympus ship big-endian containers.
|
||||
let mut v = Vec::new();
|
||||
v.extend_from_slice(b"MM");
|
||||
v.extend_from_slice(&42u16.to_be_bytes());
|
||||
v.extend_from_slice(&8u32.to_be_bytes());
|
||||
v.extend_from_slice(&2u16.to_be_bytes());
|
||||
for (tag, kind, count, value) in [
|
||||
(tag::JPEG_INTERCHANGE_FORMAT, 4u16, 1u32, 4096u32),
|
||||
(tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 900_000),
|
||||
] {
|
||||
v.extend_from_slice(&tag.to_be_bytes());
|
||||
v.extend_from_slice(&kind.to_be_bytes());
|
||||
v.extend_from_slice(&count.to_be_bytes());
|
||||
v.extend_from_slice(&value.to_be_bytes());
|
||||
}
|
||||
v.extend_from_slice(&0u32.to_be_bytes());
|
||||
v.resize(1024, 0);
|
||||
|
||||
let loc = locate_preview(&v, 25_000_000).expect("a preview");
|
||||
assert_eq!(loc.range, 4096..904_096);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_big_endian_short_reads_from_the_high_half() {
|
||||
// The classic TIFF trap: a SHORT is left-justified in the 4-byte value
|
||||
// field on big-endian, so reading it as a LONG yields a huge number.
|
||||
let mut v = Vec::new();
|
||||
v.extend_from_slice(b"MM");
|
||||
v.extend_from_slice(&42u16.to_be_bytes());
|
||||
v.extend_from_slice(&8u32.to_be_bytes());
|
||||
v.extend_from_slice(&4u16.to_be_bytes());
|
||||
for (tag, kind, count, value) in [
|
||||
(tag::STRIP_OFFSETS, 4u16, 1u32, 1000u32),
|
||||
(tag::STRIP_BYTE_COUNTS, 4, 1, 500_000),
|
||||
(tag::COMPRESSION, 3, 1, (COMPRESSION_JPEG) << 16),
|
||||
(tag::IMAGE_WIDTH, 3, 1, 1620u32 << 16),
|
||||
] {
|
||||
v.extend_from_slice(&tag.to_be_bytes());
|
||||
v.extend_from_slice(&kind.to_be_bytes());
|
||||
v.extend_from_slice(&count.to_be_bytes());
|
||||
v.extend_from_slice(&value.to_be_bytes());
|
||||
}
|
||||
v.extend_from_slice(&0u32.to_be_bytes());
|
||||
v.resize(1024, 0);
|
||||
|
||||
let loc = locate_preview(&v, 25_000_000).expect("a preview");
|
||||
assert_eq!(loc.width, Some(1620), "short read from the wrong half");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_largest_preview_wins_across_ifds() {
|
||||
// Cameras carry both a tiny thumbnail and a screen-sized preview; the
|
||||
// larger downscales better.
|
||||
let mut v = tiff(
|
||||
&[
|
||||
(tag::JPEG_INTERCHANGE_FORMAT, 4, 1, 1000),
|
||||
(tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 8_000),
|
||||
],
|
||||
200,
|
||||
);
|
||||
// A second IFD at offset 200 with a much larger preview.
|
||||
let second = 200usize;
|
||||
v[second..second + 2].copy_from_slice(&2u16.to_le_bytes());
|
||||
for (i, (tag, kind, count, value)) in [
|
||||
(tag::JPEG_INTERCHANGE_FORMAT, 4u16, 1u32, 20_000u32),
|
||||
(tag::JPEG_INTERCHANGE_FORMAT_LENGTH, 4, 1, 1_200_000),
|
||||
]
|
||||
.iter()
|
||||
.enumerate()
|
||||
{
|
||||
let e = second + 2 + i * 12;
|
||||
v[e..e + 2].copy_from_slice(&tag.to_le_bytes());
|
||||
v[e + 2..e + 4].copy_from_slice(&kind.to_le_bytes());
|
||||
v[e + 4..e + 8].copy_from_slice(&count.to_le_bytes());
|
||||
v[e + 8..e + 12].copy_from_slice(&value.to_le_bytes());
|
||||
}
|
||||
|
||||
let loc = locate_preview(&v, 25_000_000).expect("a preview");
|
||||
assert_eq!(loc.len(), 1_200_000, "should pick the larger");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_self_referential_ifd_chain_terminates() {
|
||||
// A malformed file pointing an IFD at itself must not hang the app.
|
||||
let h = tiff(&[(tag::IMAGE_WIDTH, 4, 1, 100)], 8);
|
||||
let _ = locate_preview(&h, 1_000_000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_absurd_entry_count_is_rejected_not_allocated() {
|
||||
let mut v = Vec::new();
|
||||
v.extend_from_slice(b"II");
|
||||
v.extend_from_slice(&42u16.to_le_bytes());
|
||||
v.extend_from_slice(&8u32.to_le_bytes());
|
||||
v.extend_from_slice(&60000u16.to_le_bytes()); // claims 60k entries
|
||||
v.resize(1024, 0);
|
||||
assert!(locate_preview(&v, 1_000_000).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_tiff_input_declines_cleanly() {
|
||||
assert!(locate_preview(b"not a tiff at all", 1000).is_none());
|
||||
assert!(locate_preview(&[], 1000).is_none());
|
||||
// CR3 is ISO-BMFF, not TIFF — declining is correct.
|
||||
assert!(locate_preview(b"\0\0\0\x18ftypcrx ", 1000).is_none());
|
||||
}
|
||||
|
||||
|
||||
/// Build a JPEG carrying an APP1 EXIF block with the given IFD entries.
|
||||
fn jpeg_with_exif(entries: &[(u16, u16, u32, u32)], extra: &[u8]) -> Vec<u8> {
|
||||
let mut tiff = Vec::new();
|
||||
tiff.extend_from_slice(b"II");
|
||||
tiff.extend_from_slice(&42u16.to_le_bytes());
|
||||
tiff.extend_from_slice(&8u32.to_le_bytes());
|
||||
tiff.extend_from_slice(&(entries.len() as u16).to_le_bytes());
|
||||
for (tag, kind, count, value) in entries {
|
||||
tiff.extend_from_slice(&tag.to_le_bytes());
|
||||
tiff.extend_from_slice(&kind.to_le_bytes());
|
||||
tiff.extend_from_slice(&count.to_le_bytes());
|
||||
tiff.extend_from_slice(&value.to_le_bytes());
|
||||
}
|
||||
tiff.extend_from_slice(&0u32.to_le_bytes());
|
||||
tiff.extend_from_slice(extra);
|
||||
|
||||
let payload_len = (tiff.len() + 6 + 2) as u16;
|
||||
let mut out = vec![0xFF, 0xD8, 0xFF, 0xE1];
|
||||
out.extend_from_slice(&payload_len.to_be_bytes());
|
||||
out.extend_from_slice(b"Exif\0\0");
|
||||
out.extend_from_slice(&tiff);
|
||||
out
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn jpeg_exif_yields_a_capture_time() {
|
||||
// rawler decodes no JPEG at all, so without this path every JPEG in a
|
||||
// library is undated — 5,712 scanned frames in the reference library.
|
||||
let date = b"2013:06:28 23:32:54\0";
|
||||
let mut extra = Vec::new();
|
||||
let date_offset = 8 + 2 + 12 + 4;
|
||||
extra.extend_from_slice(date);
|
||||
|
||||
let jpeg = jpeg_with_exif(
|
||||
&[(
|
||||
exif_tag::DATE_TIME_ORIGINAL,
|
||||
2,
|
||||
date.len() as u32,
|
||||
date_offset as u32,
|
||||
)],
|
||||
&extra,
|
||||
);
|
||||
|
||||
let md = jpeg_metadata(&jpeg).expect("EXIF");
|
||||
assert_eq!(md.captured_at, Some(1_372_462_374));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_jpeg_without_exif_reports_no_segment() {
|
||||
assert!(jpeg_metadata(&[0xFF, 0xD8, 0xFF, 0xDA, 0, 2]).is_err());
|
||||
assert!(jpeg_metadata(b"not a jpeg").is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_values_lose_their_nul_padding() {
|
||||
// A model name with a trailing NUL compares unequal to the same name
|
||||
// without one, which would split one camera into two in any grouping.
|
||||
let model = b"CanoScan 9000F Mark II\0";
|
||||
let mut extra = Vec::new();
|
||||
let off = 8 + 2 + 12 + 4;
|
||||
extra.extend_from_slice(model);
|
||||
|
||||
let jpeg = jpeg_with_exif(
|
||||
&[(exif_tag::MODEL, 2, model.len() as u32, off as u32)],
|
||||
&extra,
|
||||
);
|
||||
let md = jpeg_metadata(&jpeg).expect("EXIF");
|
||||
assert_eq!(md.model.as_deref(), Some("CanoScan 9000F Mark II"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_marker_walk_does_not_run_off_a_truncated_file() {
|
||||
// Untrusted input (NFR-SEC-1): a length field claiming more than the
|
||||
// file holds must not read past the end.
|
||||
let mut jpeg = vec![0xFF, 0xD8, 0xFF, 0xE1];
|
||||
jpeg.extend_from_slice(&60000u16.to_be_bytes());
|
||||
jpeg.extend_from_slice(b"Exif\0\0");
|
||||
assert!(jpeg_metadata(&jpeg).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn truncated_jpegs_are_detected() {
|
||||
// The bug this whole module exists to fix: a short read decodes to a
|
||||
// partial image rather than failing, so it must be caught by
|
||||
// inspection.
|
||||
let mut complete = vec![0xFF, 0xD8];
|
||||
complete.extend_from_slice(&[0x00; 200]);
|
||||
complete.extend_from_slice(&[0xFF, 0xD9]);
|
||||
assert!(is_complete_jpeg(&complete));
|
||||
|
||||
let truncated = &complete[..complete.len() - 2];
|
||||
assert!(!is_complete_jpeg(truncated));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_jpeg_bytes_are_not_complete_jpegs() {
|
||||
assert!(!is_complete_jpeg(&[]));
|
||||
assert!(!is_complete_jpeg(&[0xFF, 0xD9]));
|
||||
assert!(!is_complete_jpeg(b"PNG\r\n"));
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user