Date a DNG whose IFDs follow its pixels: read the head and the tail

The scan reads the first 256 KB of a file for its metadata. A camera
writes its IFDs at the front, so that is the whole structure; the linear
DNG a merge writes puts its first IFD after the pixels, and rawler,
given the head alone, finds no decoder in it. The composite was
catalogued without a date and sorted to the very end of the grid, after
every dated photograph — which is where a panorama merged on the tablet
went unfound.

dr-decode's own TIFF reader now reads through a head and a tail at a
known offset; trailing_ifd says where the tail starts and
metadata_split reads the two together. The scan, when the head fails
and points beyond itself, fetches from the IFD to the end — kilobytes —
and dates the file from both. Tested against the writer's own output.
This commit is contained in:
2026-09-19 22:35:58 +02:00
parent 9d04ff2154
commit 48c4b403d2
5 changed files with 193 additions and 47 deletions
+20
View File
@@ -304,6 +304,26 @@ pub fn metadata(bytes: &[u8]) -> Result<Metadata, DecodeError> {
error::guarded("metadata", || metadata_unguarded(bytes))
}
/// TRACES: FR-CAT-5
/// Where a TIFF-shaped file keeps its first IFD, when the head handed to
/// [`metadata`] does not reach it — the linear DNG a merge writes puts its
/// IFDs after the pixels, and rawler, given the head alone, finds no
/// decoder in it. The caller fetches from this offset to the end and
/// reads the two ranges with [`metadata_split`].
pub fn trailing_ifd(head: &[u8]) -> Option<u64> {
locate::trailing_ifd(head)
}
/// TRACES: FR-CAT-5
/// [`metadata`] for a file read in two ranges: `head` from offset 0 and
/// `tail` from `tail_at`. The EXIF sub-IFD such a file wrote before its
/// pixels is in the head; the first IFD and its values are in the tail.
pub fn metadata_split(head: &[u8], tail: &[u8], tail_at: u64) -> Result<Metadata, DecodeError> {
error::guarded("metadata", || {
locate::tiff_metadata_split(head, tail, tail_at)
})
}
fn metadata_unguarded(bytes: &[u8]) -> Result<Metadata, DecodeError> {
use rawler::rawsource::RawSource;
+82 -13
View File
@@ -183,22 +183,65 @@ struct Entry {
value: u32,
}
/// The bytes a [`TiffReader`] reads: the head of a file, and optionally a
/// second range from further in, at a known offset.
///
/// A camera writes its IFDs at the front, so the first 256 KB of a file
/// is the whole structure. A file written strip by strip — the linear
/// DNG a merge produces — has its first IFD at the *end*, after the
/// pixels, and a reader that only has the head sees a pointer into
/// nothing. Rather than fetch 800 MB to read a date, the caller fetches
/// the head, asks [`crate::trailing_ifd`] where the IFD is, fetches that
/// tail, and reads through both. Offsets are the file's own throughout;
/// a read that falls in neither range is simply absent.
#[derive(Clone, Copy)]
struct Src<'a> {
head: &'a [u8],
tail: &'a [u8],
/// Where `tail` starts in the file.
tail_at: usize,
}
impl<'a> Src<'a> {
fn whole(data: &'a [u8]) -> Self {
Src {
head: data,
tail: &[],
tail_at: 0,
}
}
fn get(&self, start: usize, len: usize) -> Option<&'a [u8]> {
let end = start.checked_add(len)?;
if let Some(b) = self.head.get(start..end) {
return Some(b);
}
let s = start.checked_sub(self.tail_at)?;
self.tail.get(s..s.checked_add(len)?)
}
}
/// A minimal TIFF structure reader.
///
/// Deliberately not a general TIFF parser: it reads the IFD chain and entry
/// values and nothing else, because that is all locating a preview needs.
struct TiffReader<'a> {
data: &'a [u8],
data: Src<'a>,
little_endian: bool,
first_ifd: u32,
}
impl<'a> TiffReader<'a> {
fn new(data: &'a [u8]) -> Option<Self> {
if data.len() < 8 {
Self::over(Src::whole(data))
}
fn over(data: Src<'a>) -> Option<Self> {
let head = data.head;
if head.len() < 8 {
return None;
}
let little_endian = match &data[0..2] {
let little_endian = match &head[0..2] {
b"II" => true,
b"MM" => false,
_ => return None,
@@ -362,9 +405,7 @@ impl<'a> TiffReader<'a> {
};
raw[..len.min(4)].to_vec()
} else {
self.data
.get(e.value as usize..e.value as usize + len)?
.to_vec()
self.data.get(e.value as usize, len)?.to_vec()
};
let s = String::from_utf8_lossy(&bytes);
@@ -396,8 +437,7 @@ impl<'a> TiffReader<'a> {
// same way to recover the original byte order.
return None;
}
let start = e.value as usize;
self.data.get(start..start.checked_add(len)?)
self.data.get(e.value as usize, len)
}
fn offsets(&self, e: &Entry) -> Vec<u32> {
@@ -420,8 +460,8 @@ impl<'a> TiffReader<'a> {
}
}
fn read_u16(data: &[u8], at: usize, le: bool) -> Option<u16> {
let b = data.get(at..at + 2)?;
fn read_u16(data: Src<'_>, at: usize, le: bool) -> Option<u16> {
let b = data.get(at, 2)?;
Some(if le {
u16::from_le_bytes([b[0], b[1]])
} else {
@@ -429,8 +469,8 @@ fn read_u16(data: &[u8], at: usize, le: bool) -> Option<u16> {
})
}
fn read_u32(data: &[u8], at: usize, le: bool) -> Option<u32> {
let b = data.get(at..at + 4)?;
fn read_u32(data: Src<'_>, at: usize, le: bool) -> Option<u32> {
let b = data.get(at, 4)?;
Some(if le {
u32::from_le_bytes([b[0], b[1], b[2], b[3]])
} else {
@@ -460,7 +500,36 @@ pub fn jpeg_metadata(bytes: &[u8]) -> Result<crate::Metadata, crate::DecodeError
/// no `DateTimeOriginal` for some DNGs whose tag sits plainly at byte 826 —
/// and without this fallback those images are silently undated.
pub fn tiff_metadata(tiff_data: &[u8]) -> Result<crate::Metadata, crate::DecodeError> {
let reader = TiffReader::new(tiff_data)
tiff_metadata_over(Src::whole(tiff_data))
}
/// Where a TIFF-shaped file's first IFD is, when the head does not reach
/// it: the offset to fetch from, so [`tiff_metadata_split`] can read it.
/// `None` for a file that is not TIFF, or whose IFD the head already holds.
pub fn trailing_ifd(head: &[u8]) -> Option<u64> {
let r = TiffReader::new(head)?;
let at = r.first_ifd as u64;
(at >= head.len() as u64).then_some(at)
}
/// [`tiff_metadata`] over a head and a tail fetched separately: the head
/// from offset 0, the tail from `tail_at`. For the file whose IFDs follow
/// its pixels.
pub fn tiff_metadata_split(
head: &[u8],
tail: &[u8],
tail_at: u64,
) -> Result<crate::Metadata, crate::DecodeError> {
tiff_metadata_over(Src {
head,
tail,
tail_at: usize::try_from(tail_at)
.map_err(|_| crate::DecodeError::Metadata("tail offset out of range".into()))?,
})
}
fn tiff_metadata_over(src: Src<'_>) -> Result<crate::Metadata, crate::DecodeError> {
let reader = TiffReader::over(src)
.ok_or_else(|| crate::DecodeError::Metadata("malformed EXIF header".into()))?;
let mut md = crate::Metadata::default();
+28
View File
@@ -241,6 +241,8 @@ mod tests {
let source = SourceMetadata {
make: Some("Canon".into()),
model: Some("Canon EOS 6D".into()),
captured_at: Some(1_754_398_664),
captured_offset: Some(120),
..Default::default()
};
write_linear_dng(
@@ -301,6 +303,32 @@ mod tests {
assert_eq!((crop.p.x, crop.p.y, crop.d.w, crop.d.h), (2, 1, 15, 10));
}
#[test]
fn the_catalog_reads_the_date_from_a_head_and_a_tail() {
// TRACES: FR-CAT-5
// The IFDs follow the pixels, so a scan that has the first bytes of
// the file has a pointer into nothing; rawler finds no decoder in
// that, and the composite would sit undated at the end of the grid.
// The scan's second range — from the first IFD to the end — with
// the head is enough to date it, and to name the camera.
let bytes = write(640, 400, 64);
let head = &bytes[..4096];
assert!(
dr_decode::metadata(head).is_err(),
"the head alone must not read"
);
let at = dr_decode::trailing_ifd(head).expect("the IFD is beyond the head");
assert!(at as usize > head.len());
let tail = &bytes[at as usize..];
assert!(tail.len() < 4096, "the tail is the IFD, not the pixels");
let md = dr_decode::metadata_split(head, tail, at).expect("read from two ranges");
assert_eq!(md.captured_at, Some(1_754_398_664));
assert_eq!(md.captured_offset, Some(120));
assert_eq!(md.model.as_deref(), Some("Canon EOS 6D"));
// A head that holds everything is not a trailing-IFD file.
assert_eq!(dr_decode::trailing_ifd(&bytes), None);
}
#[test]
fn a_strip_of_the_wrong_length_is_refused() {
let mut bytes = std::io::Cursor::new(Vec::new());
+29 -29
View File
File diff suppressed because one or more lines are too long
+33 -4
View File
@@ -3230,7 +3230,7 @@ async fn fetch_preview(
// The same bytes carry EXIF. Reading it here is free — the alternative is
// a second 256 KB fetch per image over the whole library.
if req.needs_metadata {
collect_metadata(&header, req, found_metadata);
collect_metadata(backend, &id, &header, req, found_metadata).await;
}
// Read unconditionally, unlike the rest of the EXIF above: `needs_metadata`
@@ -3350,10 +3350,39 @@ pub(crate) fn store_thumbnail(
///
/// Shared by both paths — the thumbnail fetch, which gets the header anyway,
/// and the header-only pass for images whose pixels were already cached.
fn collect_metadata(header: &[u8], req: &ThumbnailRequest, out: &mut Vec<MetadataFound>) {
let Ok(md) = dr_decode::metadata(header) else {
async fn collect_metadata(
backend: &dyn RemoteBackend,
id: &RemoteId,
header: &[u8],
req: &ThumbnailRequest,
out: &mut Vec<MetadataFound>,
) {
let md = match dr_decode::metadata(header) {
Ok(md) => md,
Err(first) => {
// A file whose IFDs follow its pixels — the linear DNG a merge
// writes — has nothing for the decoder in its head but a
// pointer. Its structure is a few kilobytes at the end; fetch
// that and read the two ranges together, rather than leave the
// composite undated at the end of the grid.
let Some(at) = dr_decode::trailing_ifd(header).filter(|at| *at < req.size) else {
return;
};
match backend.get(id, Some(at..req.size)).await {
Ok(tail) => match dr_decode::metadata_split(header, &tail, at) {
Ok(md) => md,
Err(e) => {
log::debug!("metadata: {}: head {first}; head and tail {e}", req.path);
return;
}
},
Err(e) => {
log::debug!("metadata: {}: tail not fetched: {e}", req.path);
return;
}
}
}
};
out.push(MetadataFound {
image_id: req.image_id,
captured_at: md.captured_at,
@@ -3442,7 +3471,7 @@ async fn read_metadata_only(
for attempt in 1..=ATTEMPTS {
match backend.get(&id, Some(0..dr_decode::HEADER_BYTES)).await {
Ok(header) => {
collect_metadata(&header, req, found);
collect_metadata(backend, &id, &header, req, found).await;
return true;
}
Err(e) if e.is_transient() && attempt < ATTEMPTS => {