//! Run the scene model over a JPEG, time it, and write what it saw. //! //! Two jobs in one example because they need the same setup and answering //! either one alone leaves the other open. //! //! **Looking.** Same argument as `detect`: no unit test settles whether the //! letterbox inverse in `Scene::rasterise` is right, because an off-by-one //! produces perfectly plausible weights over slightly the wrong pixels. A sky //! mask laid over the photograph settles it in one glance. //! //! **Timing.** Every number quoted while this model was being chosen came off a //! laptop that was compiling other things at the time, which makes them upper //! bounds and nothing better. This exists so the figure that ends up in a //! document came from a quiet machine and can be reproduced on another one. //! //! ```sh //! cargo run -p dr-segment --example scene --release --features embedded-scene-model -- photo.jpg //! cargo run -p dr-segment --example scene --release -- photo.jpg out 20 \ //! models/scene/yolo26s-sem-ade20k.onnx //! ``` //! //! Writes `-.ppm` per category — the photograph darkened //! where the category is absent, so the mask is legible *against the picture it //! came from* rather than as an abstract grey field. PPM for the same reason //! the other examples use it: no encoder dependency, and every viewer reads it. //! //! And `--refined-.ppm` beside it — the same //! category after [`dr_segment::Refinement`] has cut it back to the pixels //! whose colour agrees with it, at each point of a sweep across the whole //! strictness range the application's slider offers. //! //! A sweep rather than one image, for two reasons. Whether the refinement is //! an improvement is a comparative judgement — did the flag come out of the //! sky, and is the sky still there afterwards — which a single image cannot //! answer. And the control is a slider, so the question actually being asked //! is *where to put it*, which needs the travel rather than a point on it. //! //! The percentage printed beside each is how much weight came off, which is //! the number to be suspicious of when it grows quickly. The timing beside //! that is the fit against one `apply`, and it is the measurement that decides //! whether the slider can be a live drag. //! //! Timings are reported as a median over the requested run count, with the //! first run excluded. That first pass pays for tract's lazy allocation and is //! not representative of the second image a session decodes. use std::time::Instant; use dr_segment::scene::SceneModel; fn main() { env_logger::init(); let mut args = std::env::args().skip(1); let Some(path) = args.next() else { eprintln!( "usage: scene [out-prefix] [runs] [model.onnx classes.json categories.txt]" ); eprintln!(" with --features embedded-scene-model the model arguments may be omitted"); std::process::exit(2); }; let prefix = args.next().unwrap_or_else(|| "scene".into()); let runs: usize = args .next() .and_then(|r| r.parse().ok()) .unwrap_or(10) .max(1); let (rgb, width, height) = read_jpeg(&path); println!("{path}: {width}×{height}"); let mut model = match (args.next(), args.next(), args.next()) { (Some(m), Some(c), Some(g)) => { SceneModel::from_path(m, c, g).expect("could not load the scene model") } _ => embedded(), }; // Excluded from the statistics deliberately — see the header. let warm = Instant::now(); let scene = model .analyse(&rgb, width, height) .expect("inference failed"); println!("first run: {:?} (allocation included)", warm.elapsed()); let mut times: Vec = Vec::with_capacity(runs); for _ in 0..runs { let start = Instant::now(); let _ = model .analyse(&rgb, width, height) .expect("inference failed"); times.push(start.elapsed().as_secs_f64() * 1000.0); } times.sort_by(f64::total_cmp); println!( "{runs} runs: median {:.0} ms (min {:.0}, max {:.0})", times[times.len() / 2], times[0], times[times.len() - 1], ); let (gw, gh) = scene.grid_size(); println!("logit grid: {gw}×{gh}"); println!(); // Coverage first and sorted, because on any given photograph most // categories are absent and the two or three that are not are the whole // story. let mut ranked: Vec<(usize, f32)> = (0..scene.categories().len()) .map(|k| (k, scene.coverage(k))) .collect(); ranked.sort_by(|a, b| b.1.total_cmp(&a.1)); for (k, coverage) in ranked { let name = &scene.categories()[k]; println!("{name:>14} {:5.1}%", coverage * 100.0); // A category covering essentially nothing produces a black image and a // file nobody wants; the threshold is what the scene tab would use to // decide whether to offer a slider at all. if coverage < 0.005 { continue; } let mask = scene .rasterise(k, width, height) .expect("category index came from the same Scene"); write_overlay(&format!("{prefix}-{name}.ppm"), &rgb, &mask, width, height); // The same category cut back to the pixels whose colour agrees with // it, written *beside* the coarse one rather than instead of it. The // judgement this example exists to support is comparative — is the // flag out, and is the sky still there — and it cannot be made from // one image. // // Swept rather than shown once, because the application offers this as // a slider and the question a photographer will actually ask is "where // do I put it": the whole travel of the control, at the resolution // their own eye will judge it at. let start = Instant::now(); let refinement = dr_segment::Refinement::compute( &mask, &rgb, width, height, scene.cell_pixels(), &dr_segment::RefineOptions::default(), ); let fit = start.elapsed(); let refinement = match refinement { Ok(r) => r, Err(why) => { println!(" left coarse: {why:?}"); continue; } }; println!(" fitted in {fit:?}"); let total: f32 = mask.iter().sum(); for step in 0..=STRICTNESS_STEPS { let strictness = step as f32 * dr_segment::STRICTNESS_MAX / STRICTNESS_STEPS as f32; // Timed separately, and this is the number that decides whether // the control can be a live drag or has to wait for the release. let start = Instant::now(); let refined = refinement.apply(&mask, strictness); let took = start.elapsed(); let cut = if total > 0.0 { (total - refined.iter().sum::()) / total } else { 0.0 }; println!( " strictness {strictness:>4.1}: {:5.1}% cut ({took:?})", cut * 100.0 ); write_overlay( &format!("{prefix}-{name}-refined-{strictness:.1}.ppm"), &rgb, &refined, width, height, ); } } } /// Points on the strictness sweep, over `0..=STRICTNESS_MAX`. /// /// Eight, so the files come out at whole nats from `0` to `8` — few enough to /// look at every one of them, which is the point of writing them at all, and /// spaced at the unit the control is actually denominated in. const STRICTNESS_STEPS: usize = 8; #[cfg(feature = "embedded-scene-model")] fn embedded() -> SceneModel { SceneModel::embedded().expect("could not load the embedded scene model") } #[cfg(not(feature = "embedded-scene-model"))] fn embedded() -> SceneModel { eprintln!( "no model given, and this build has no embedded one.\n\ Either pass the three paths, or rebuild with --features embedded-scene-model." ); std::process::exit(2); } /// The photograph, dimmed where the category is not. /// /// Not a bare greyscale mask: the question being asked is "does this weight /// land on the sky", and a mask on its own cannot answer it — you have to see /// the sky underneath. A floor rather than a multiply, so that a region the /// model gave up on is still visible enough to recognise. fn write_overlay(path: &str, rgb: &[f32], mask: &[f32], width: usize, height: usize) { let mut out = String::with_capacity(64); out.push_str(&format!("P3\n{width} {height}\n255\n")); let mut bytes = out.into_bytes(); for i in 0..width * height { let w = mask[i].clamp(0.0, 1.0); let gain = 0.15 + 0.85 * w; for c in 0..3 { let v = (rgb[i * 3 + c] * gain * 255.0).clamp(0.0, 255.0) as u8; bytes.extend_from_slice(v.to_string().as_bytes()); bytes.push(if c == 2 { b'\n' } else { b' ' }); } } match std::fs::write(path, bytes) { Ok(()) => println!(" wrote {path}"), Err(e) => eprintln!(" could not write {path}: {e}"), } } /// Decode to the tightly packed `f32` RGB the model wants. fn read_jpeg(path: &str) -> (Vec, usize, usize) { let bytes = std::fs::read(path).expect("could not read the photograph"); let mut decoder = zune_jpeg::JpegDecoder::new(&bytes); let pixels = decoder.decode().expect("could not decode the photograph"); let info = decoder.info().expect("decoded image has no dimensions"); let (width, height) = (info.width as usize, info.height as usize); // zune hands back whatever the file had. Three channels is the ordinary // case; one is a greyscale scan, which is worth handling because a // black-and-white frame is exactly the kind of thing someone reaches for // when a colour one looks wrong. let components = pixels.len() / (width * height); let rgb = match components { 3 => pixels.iter().map(|&p| p as f32 / 255.0).collect(), 1 => pixels.iter().flat_map(|&p| [p as f32 / 255.0; 3]).collect(), n => panic!("unsupported component count: {n}"), }; (rgb, width, height) }