Merge branch 'master' into clarity-reduced-base
# Conflicts: # docs/traceability.md
This commit is contained in:
+230
-30
@@ -74,6 +74,68 @@ pub struct SharedGpu {
|
||||
pub adapter: wgpu::Adapter,
|
||||
}
|
||||
|
||||
/// TRACES: FR-DSP-1 | NFR-RES-4
|
||||
/// Which GPU to prefer, on a machine with more than one.
|
||||
///
|
||||
/// **Not obviously the fastest one**, which is why this is a choice rather
|
||||
/// than a constant. A discrete card wins on raw compute and loses on every
|
||||
/// byte that has to reach it: a 24 MP frame is ~96 MB of RGBA, and each
|
||||
/// upload and each export readback crosses PCIe. An integrated GPU shares
|
||||
/// memory with the CPU, so those transfers are not transfers. It also does not
|
||||
/// empty a laptop battery.
|
||||
///
|
||||
/// Which of those dominates depends on the work — a slider drag over a
|
||||
/// resident texture is compute-bound and favours the discrete card, while
|
||||
/// import, export and thumbnailing are transfer-heavy — so the honest thing is
|
||||
/// to let it be set rather than to assume.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
|
||||
pub enum AdapterPreference {
|
||||
/// The most capable GPU. What this has always done, and the default: it is
|
||||
/// the right answer for interactive editing, which is the frame budget
|
||||
/// that FR-DSP-3 actually measures.
|
||||
#[default]
|
||||
Performance,
|
||||
/// An integrated GPU where there is one — shared memory, no bus crossing,
|
||||
/// and far less power.
|
||||
Efficiency,
|
||||
}
|
||||
|
||||
impl AdapterPreference {
|
||||
/// Read the override, defaulting to [`Performance`](Self::Performance).
|
||||
///
|
||||
/// An environment variable rather than a setting, *for now*: this belongs
|
||||
/// on the settings page beside the cache budget, and putting it there
|
||||
/// needs a control and a restart prompt, because the device is opened once
|
||||
/// at startup and shared with the compositor. The variable is what makes
|
||||
/// the choice testable and gives someone with a broken primary GPU a way
|
||||
/// out today.
|
||||
pub fn from_env() -> Self {
|
||||
match std::env::var("DARKROOM_GPU").as_deref() {
|
||||
Ok("integrated") | Ok("efficiency") | Ok("igpu") => Self::Efficiency,
|
||||
_ => Self::Performance,
|
||||
}
|
||||
}
|
||||
|
||||
/// How much we want an adapter, lowest first.
|
||||
///
|
||||
/// A CPU adapter sorts last under both policies rather than being
|
||||
/// excluded: software rendering is a poor experience and a working one,
|
||||
/// and on a machine where every real GPU has failed it is the difference
|
||||
/// between a slow editor and no editor.
|
||||
fn rank(self, device_type: wgpu::DeviceType) -> u8 {
|
||||
use wgpu::DeviceType as D;
|
||||
match (self, device_type) {
|
||||
(_, D::Cpu) => 4,
|
||||
(Self::Performance, D::DiscreteGpu) => 0,
|
||||
(Self::Performance, D::IntegratedGpu) => 1,
|
||||
(Self::Efficiency, D::IntegratedGpu) => 0,
|
||||
(Self::Efficiency, D::DiscreteGpu) => 1,
|
||||
(_, D::VirtualGpu) => 2,
|
||||
(_, D::Other) => 3,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl GpuContext {
|
||||
/// Create a headless context — no surface, no window.
|
||||
///
|
||||
@@ -111,6 +173,38 @@ impl GpuContext {
|
||||
Self::open(wgpu::Backends::VULKAN).await
|
||||
}
|
||||
|
||||
/// TRACES: FR-DSP-1 | NFR-R1
|
||||
/// Open a device, trying every adapter rather than only the best one.
|
||||
///
|
||||
/// # Why this is not `request_adapter`
|
||||
///
|
||||
/// `request_adapter` with `HighPerformance` returns *one* adapter and no
|
||||
/// second chance. That is the right answer on a healthy machine and the
|
||||
/// wrong one on a machine with a sick GPU, which is not a rare state:
|
||||
/// observed 2026-08-29 on a laptop whose discrete card had hit an NVRM
|
||||
/// assertion failure and a fullchip reset. The driver still advertised the
|
||||
/// adapter, `request_adapter` dutifully picked it as the highest
|
||||
/// performing, and the process died on it — while a working integrated GPU
|
||||
/// and a working external card sat unused in the same enumeration.
|
||||
///
|
||||
/// A photo editor that will not start because the *fastest* GPU is broken,
|
||||
/// on a machine holding two that are not, is worse than a slow one.
|
||||
///
|
||||
/// So: enumerate, order by how much we want each, and take the first that
|
||||
/// actually yields a device. The ordering reproduces what
|
||||
/// `HighPerformance` means — discrete, then integrated, then anything —
|
||||
/// so the healthy case picks exactly what it picked before and pays one
|
||||
/// extra enumeration for it.
|
||||
///
|
||||
/// # What this cannot do
|
||||
///
|
||||
/// A GPU sick enough to accept `request_device` and fail later is still
|
||||
/// fatal, because the failure arrives as a segfault inside the driver
|
||||
/// rather than as an error we could catch. This moves the boundary from
|
||||
/// "the preferred adapter is unusable" to "the preferred adapter is
|
||||
/// unusable *and* dishonest about it"; it does not remove it. Device loss
|
||||
/// after a successful open is a different problem with a different answer
|
||||
/// (ARCH §5.6).
|
||||
async fn open(backends: wgpu::Backends) -> Result<SharedGpu, GpuError> {
|
||||
// `new_without_display_handle` rather than a struct literal: the
|
||||
// descriptor carries a boxed display handle and so has no `Default`,
|
||||
@@ -119,27 +213,61 @@ impl GpuContext {
|
||||
descriptor.backends = backends;
|
||||
let instance = wgpu::Instance::new(descriptor);
|
||||
|
||||
let adapter = instance
|
||||
.request_adapter(&wgpu::RequestAdapterOptions {
|
||||
power_preference: wgpu::PowerPreference::HighPerformance,
|
||||
compatible_surface: None,
|
||||
force_fallback_adapter: false,
|
||||
})
|
||||
.await
|
||||
// A `Result` since wgpu 24, where it was an `Option`. The error
|
||||
// says which backends were tried, which is worth more than the
|
||||
// bare "no adapter" this used to report.
|
||||
.map_err(|_| GpuError::NoAdapter)?;
|
||||
let mut adapters: Vec<wgpu::Adapter> = instance.enumerate_adapters(backends).await;
|
||||
if adapters.is_empty() {
|
||||
return Err(GpuError::NoAdapter);
|
||||
}
|
||||
let policy = AdapterPreference::from_env();
|
||||
adapters.sort_by_key(|a| policy.rank(a.get_info().device_type));
|
||||
|
||||
let adapter_info = adapter.get_info();
|
||||
log::info!(
|
||||
"gpu: {} ({:?}, {:?})",
|
||||
adapter_info.name,
|
||||
adapter_info.device_type,
|
||||
adapter_info.backend
|
||||
);
|
||||
// Kept so a total failure can say what it tried. "No suitable GPU
|
||||
// adapter found" on a machine with three of them sends the reader to
|
||||
// look for a driver that is installed and loaded.
|
||||
let mut refusals: Vec<String> = Vec::new();
|
||||
|
||||
let (device, queue) = adapter
|
||||
for adapter in adapters {
|
||||
let adapter_info = adapter.get_info();
|
||||
match Self::device_from(&adapter).await {
|
||||
Ok((device, queue)) => {
|
||||
log::info!(
|
||||
"gpu: {} ({:?}, {:?})",
|
||||
adapter_info.name,
|
||||
adapter_info.device_type,
|
||||
adapter_info.backend
|
||||
);
|
||||
if !refusals.is_empty() {
|
||||
// At `info`, not `debug`: the user is now running on
|
||||
// their second-choice GPU and any performance
|
||||
// complaint that follows begins here.
|
||||
log::info!(
|
||||
"gpu: fell back after {} unusable adapter(s): {}",
|
||||
refusals.len(),
|
||||
refusals.join("; ")
|
||||
);
|
||||
}
|
||||
return Ok(SharedGpu {
|
||||
ctx: Self {
|
||||
device: Arc::new(device),
|
||||
queue: Arc::new(queue),
|
||||
adapter_info,
|
||||
},
|
||||
instance,
|
||||
adapter,
|
||||
});
|
||||
}
|
||||
Err(e) => refusals.push(format!("{} ({e})", adapter_info.name)),
|
||||
}
|
||||
}
|
||||
|
||||
Err(GpuError::DeviceRequest(format!(
|
||||
"every adapter refused a device: {}",
|
||||
refusals.join("; ")
|
||||
)))
|
||||
}
|
||||
|
||||
/// Ask one adapter for a device, with the limits the pipeline needs.
|
||||
async fn device_from(adapter: &wgpu::Adapter) -> Result<(wgpu::Device, wgpu::Queue), GpuError> {
|
||||
adapter
|
||||
.request_device(&wgpu::DeviceDescriptor {
|
||||
label: Some("darkroom-device"),
|
||||
required_features: wgpu::Features::empty(),
|
||||
@@ -164,17 +292,7 @@ impl GpuContext {
|
||||
trace: wgpu::Trace::Off,
|
||||
})
|
||||
.await
|
||||
.map_err(|e| GpuError::DeviceRequest(e.to_string()))?;
|
||||
|
||||
Ok(SharedGpu {
|
||||
ctx: Self {
|
||||
device: Arc::new(device),
|
||||
queue: Arc::new(queue),
|
||||
adapter_info,
|
||||
},
|
||||
instance,
|
||||
adapter,
|
||||
})
|
||||
.map_err(|e| GpuError::DeviceRequest(e.to_string()))
|
||||
}
|
||||
|
||||
/// Build a context from a device and queue owned by someone else — the
|
||||
@@ -597,3 +715,85 @@ mod tests {
|
||||
assert_eq!(rt.size(), (1, 1));
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod adapter_choice_tests {
|
||||
//! Which GPU gets picked, and what happens when it will not open.
|
||||
//!
|
||||
//! These are about the *ordering*, which is pure — opening a device needs
|
||||
//! hardware and is covered by every other test in this crate implicitly.
|
||||
|
||||
use super::*;
|
||||
|
||||
/// Only the two fields the ordering reads are set; the rest come from
|
||||
/// `Default`, so a wgpu upgrade that adds another does not break this.
|
||||
/// The order adapters would be tried in, named so a failure reads as the
|
||||
/// hardware it stands for.
|
||||
fn order(policy: AdapterPreference, mut gpus: Vec<(&str, wgpu::DeviceType)>) -> Vec<&str> {
|
||||
gpus.sort_by_key(|(_, t)| policy.rank(*t));
|
||||
gpus.into_iter().map(|(name, _)| name).collect()
|
||||
}
|
||||
|
||||
fn a_laptop() -> Vec<(&'static str, wgpu::DeviceType)> {
|
||||
vec![
|
||||
("Iris Xe", wgpu::DeviceType::IntegratedGpu),
|
||||
("RTX 3050", wgpu::DeviceType::DiscreteGpu),
|
||||
("llvmpipe", wgpu::DeviceType::Cpu),
|
||||
]
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn performance_takes_the_discrete_card() {
|
||||
// What this has always done, and what an interactive slider drag wants:
|
||||
// the texture is already resident, so the work is compute and the bus
|
||||
// does not come into it.
|
||||
assert_eq!(
|
||||
order(AdapterPreference::Performance, a_laptop()),
|
||||
["RTX 3050", "Iris Xe", "llvmpipe"]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn efficiency_takes_the_integrated_one() {
|
||||
// Shared memory, so a 96 MB frame upload is not a transfer, and a
|
||||
// laptop battery that lasts. The discrete card stays as the fallback
|
||||
// rather than being excluded.
|
||||
assert_eq!(
|
||||
order(AdapterPreference::Efficiency, a_laptop()),
|
||||
["Iris Xe", "RTX 3050", "llvmpipe"]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn software_rendering_is_last_but_never_dropped() {
|
||||
// On a machine where every real GPU has failed this is the difference
|
||||
// between a slow editor and no editor.
|
||||
for policy in [
|
||||
AdapterPreference::Performance,
|
||||
AdapterPreference::Efficiency,
|
||||
] {
|
||||
assert_eq!(
|
||||
*order(policy, a_laptop()).last().unwrap(),
|
||||
"llvmpipe",
|
||||
"{policy:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_machine_with_one_gpu_is_unaffected_by_the_policy() {
|
||||
// The common case: no choice to make, and no behaviour to change.
|
||||
let one = vec![("Iris Xe", wgpu::DeviceType::IntegratedGpu)];
|
||||
assert_eq!(
|
||||
order(AdapterPreference::Performance, one.clone()),
|
||||
order(AdapterPreference::Efficiency, one)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_default_is_what_it_did_before() {
|
||||
// Changing which GPU an existing user lands on is not something to do
|
||||
// by accident.
|
||||
assert_eq!(AdapterPreference::default(), AdapterPreference::Performance);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user