//! Run the scene model over a JPEG, time it, and write what it saw. //! //! Two jobs in one example because they need the same setup and answering //! either one alone leaves the other open. //! //! **Looking.** Same argument as `detect`: no unit test settles whether the //! letterbox inverse in `Scene::rasterise` is right, because an off-by-one //! produces perfectly plausible weights over slightly the wrong pixels. A sky //! mask laid over the photograph settles it in one glance. //! //! **Timing.** Every number quoted while this model was being chosen came off a //! laptop that was compiling other things at the time, which makes them upper //! bounds and nothing better. This exists so the figure that ends up in a //! document came from a quiet machine and can be reproduced on another one. //! //! ```sh //! cargo run -p dr-segment --example scene --release --features embedded-scene-model -- photo.jpg //! cargo run -p dr-segment --example scene --release -- photo.jpg out 20 \ //! models/scene/yolo26s-sem-ade20k.onnx //! ``` //! //! Writes `-.ppm` per category — the photograph darkened //! where the category is absent, so the mask is legible *against the picture it //! came from* rather than as an abstract grey field. PPM for the same reason //! the other examples use it: no encoder dependency, and every viewer reads it. //! //! And `--refined.ppm` beside it, which is the same category //! after [`dr_segment::refine_category`] has cut it back to the pixels whose //! colour agrees with it. Both, never one: whether that refinement is an //! improvement is a comparative judgement — did the flag come out of the sky, //! and is the sky still there afterwards — and a single image cannot answer //! it. The percentage printed beside each is how much weight came off, which //! is the number to be suspicious of when it is large. //! //! Timings are reported as a median over the requested run count, with the //! first run excluded. That first pass pays for tract's lazy allocation and is //! not representative of the second image a session decodes. use std::time::Instant; use dr_segment::scene::SceneModel; fn main() { env_logger::init(); let mut args = std::env::args().skip(1); let Some(path) = args.next() else { eprintln!( "usage: scene [out-prefix] [runs] [model.onnx classes.json categories.txt]" ); eprintln!(" with --features embedded-scene-model the model arguments may be omitted"); std::process::exit(2); }; let prefix = args.next().unwrap_or_else(|| "scene".into()); let runs: usize = args .next() .and_then(|r| r.parse().ok()) .unwrap_or(10) .max(1); let (rgb, width, height) = read_jpeg(&path); println!("{path}: {width}×{height}"); let mut model = match (args.next(), args.next(), args.next()) { (Some(m), Some(c), Some(g)) => { SceneModel::from_path(m, c, g).expect("could not load the scene model") } _ => embedded(), }; // Excluded from the statistics deliberately — see the header. let warm = Instant::now(); let scene = model .analyse(&rgb, width, height) .expect("inference failed"); println!("first run: {:?} (allocation included)", warm.elapsed()); let mut times: Vec = Vec::with_capacity(runs); for _ in 0..runs { let start = Instant::now(); let _ = model .analyse(&rgb, width, height) .expect("inference failed"); times.push(start.elapsed().as_secs_f64() * 1000.0); } times.sort_by(f64::total_cmp); println!( "{runs} runs: median {:.0} ms (min {:.0}, max {:.0})", times[times.len() / 2], times[0], times[times.len() - 1], ); let (gw, gh) = scene.grid_size(); println!("logit grid: {gw}×{gh}"); println!(); // Coverage first and sorted, because on any given photograph most // categories are absent and the two or three that are not are the whole // story. let mut ranked: Vec<(usize, f32)> = (0..scene.categories().len()) .map(|k| (k, scene.coverage(k))) .collect(); ranked.sort_by(|a, b| b.1.total_cmp(&a.1)); for (k, coverage) in ranked { let name = &scene.categories()[k]; println!("{name:>14} {:5.1}%", coverage * 100.0); // A category covering essentially nothing produces a black image and a // file nobody wants; the threshold is what the scene tab would use to // decide whether to offer a slider at all. if coverage < 0.005 { continue; } let mask = scene .rasterise(k, width, height) .expect("category index came from the same Scene"); write_overlay(&format!("{prefix}-{name}.ppm"), &rgb, &mask, width, height); // The same category cut back to the pixels whose colour agrees with // it, written *beside* the coarse one rather than instead of it. The // judgement this example exists to support is comparative — is the // flag out, and is the sky still there — and it cannot be made from // one image. let start = Instant::now(); let (refined, what) = dr_segment::refine_category( &mask, &rgb, width, height, scene.cell_pixels(), &dr_segment::RefineOptions::default(), ); let took = start.elapsed(); match what { dr_segment::Refined::Applied { removed } => { println!( " refined in {took:?}: {:.1}% of the weight cut", removed * 100.0 ); write_overlay( &format!("{prefix}-{name}-refined.ppm"), &rgb, &refined, width, height, ); } dr_segment::Refined::Skipped(why) => { println!(" left coarse: {why:?}"); } } } } #[cfg(feature = "embedded-scene-model")] fn embedded() -> SceneModel { SceneModel::embedded().expect("could not load the embedded scene model") } #[cfg(not(feature = "embedded-scene-model"))] fn embedded() -> SceneModel { eprintln!( "no model given, and this build has no embedded one.\n\ Either pass the three paths, or rebuild with --features embedded-scene-model." ); std::process::exit(2); } /// The photograph, dimmed where the category is not. /// /// Not a bare greyscale mask: the question being asked is "does this weight /// land on the sky", and a mask on its own cannot answer it — you have to see /// the sky underneath. A floor rather than a multiply, so that a region the /// model gave up on is still visible enough to recognise. fn write_overlay(path: &str, rgb: &[f32], mask: &[f32], width: usize, height: usize) { let mut out = String::with_capacity(64); out.push_str(&format!("P3\n{width} {height}\n255\n")); let mut bytes = out.into_bytes(); for i in 0..width * height { let w = mask[i].clamp(0.0, 1.0); let gain = 0.15 + 0.85 * w; for c in 0..3 { let v = (rgb[i * 3 + c] * gain * 255.0).clamp(0.0, 255.0) as u8; bytes.extend_from_slice(v.to_string().as_bytes()); bytes.push(if c == 2 { b'\n' } else { b' ' }); } } match std::fs::write(path, bytes) { Ok(()) => println!(" wrote {path}"), Err(e) => eprintln!(" could not write {path}: {e}"), } } /// Decode to the tightly packed `f32` RGB the model wants. fn read_jpeg(path: &str) -> (Vec, usize, usize) { let bytes = std::fs::read(path).expect("could not read the photograph"); let mut decoder = zune_jpeg::JpegDecoder::new(&bytes); let pixels = decoder.decode().expect("could not decode the photograph"); let info = decoder.info().expect("decoded image has no dimensions"); let (width, height) = (info.width as usize, info.height as usize); // zune hands back whatever the file had. Three channels is the ordinary // case; one is a greyscale scan, which is worth handling because a // black-and-white frame is exactly the kind of thing someone reaches for // when a colour one looks wrong. let components = pixels.len() / (width * height); let rgb = match components { 3 => pixels.iter().map(|&p| p as f32 / 255.0).collect(), 1 => pixels.iter().flat_map(|&p| [p as f32 / 255.0; 3]).collect(), n => panic!("unsupported component count: {n}"), }; (rgb, width, height) }