diff --git a/Cargo.lock b/Cargo.lock index 2f02550..73ab5b9 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2961,8 +2961,6 @@ dependencies = [ [[package]] name = "i-slint-renderer-skia" version = "1.17.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7b6eed7f3f0a9a3d3ca6e8b9d4ca233371d989351fdb2a7ab88ec368b99e7b57" dependencies = [ "ash", "bytemuck", @@ -7925,8 +7923,6 @@ dependencies = [ [[package]] name = "wgpu-hal" version = "29.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "97ace1c17727311c22a46e4e3faf56ea6de81af99dcc839bdfb54857b94d448d" dependencies = [ "android_system_properties", "arrayvec", diff --git a/Cargo.toml b/Cargo.toml index 590dfea..3b6f808 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -27,6 +27,9 @@ members = [ "tools/bench", "tools/traceability", ] +# Patched copies of upstream crates, not our code: see third_party/README.md. +# Excluded so `--workspace` does not test, lint or format them as ours. +exclude = ["third_party"] [workspace.package] version = "0.14.1" @@ -266,3 +269,12 @@ opt-level = 0 [profile.release] lto = "thin" codegen-units = 1 + +# Two upstream crates carry a local patch so that the Android build can draw +# with wgpu on a rotated display (technical-debt.md TD-1). Both are exact +# copies of the version the lockfile already resolves, plus that patch; +# third_party/README.md says what was changed and how to carry it forward +# when Slint or wgpu moves. +[patch.crates-io] +wgpu-hal = { path = "third_party/wgpu-hal-29.0.4" } +i-slint-renderer-skia = { path = "third_party/i-slint-renderer-skia-1.17.1" } diff --git a/third_party/README.md b/third_party/README.md new file mode 100644 index 0000000..2f9d111 --- /dev/null +++ b/third_party/README.md @@ -0,0 +1,35 @@ +# Patched upstream crates + +Each directory here is a crate exactly as crates.io publishes it, at the +version `Cargo.lock` resolves, with a local patch on top. The root +`Cargo.toml` routes the dependency here through `[patch.crates-io]`; the +directory is excluded from the workspace, so `cargo test --workspace`, +clippy and fmt leave it alone. + +The first commit that adds a directory is the pristine copy (from +`~/.cargo/registry/src/*/-`, minus `.cargo-ok` and the +crate's own `Cargo.lock`). Every later commit touching it is ours, so +`git log -p -- third_party/` is the patch and nothing else. + +## Carrying a patch forward + +When Slint or wgpu is bumped, the version here stops matching and cargo +warns that the patch is unused — the build then silently goes back to the +unpatched crate. So a bump is: + +1. Copy the new version in beside the old one, as its own commit. +2. Re-apply the diff from `git log -p` on the old directory. +3. Point `[patch.crates-io]` at the new directory and delete the old one. +4. Re-check on the device (below) — both patches are behaviour that only a + rotated Android display exercises. + +Drop a patch entirely once upstream has the fix; each section says what +upstream change would make it unnecessary. + +## wgpu-hal 29.0.4 — Vulkan pre-rotation on Android + +*(patch lands in the next commit)* + +## i-slint-renderer-skia 1.17.1 — rotate the canvas to match + +*(patch lands in a later commit)* diff --git a/third_party/i-slint-renderer-skia-1.17.1/.cargo_vcs_info.json b/third_party/i-slint-renderer-skia-1.17.1/.cargo_vcs_info.json new file mode 100644 index 0000000..52aa938 --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/.cargo_vcs_info.json @@ -0,0 +1,6 @@ +{ + "git": { + "sha1": "cf62c975c311e7036d599ed8ed0b7e6a8386a934" + }, + "path_in_vcs": "internal/renderers/skia" +} \ No newline at end of file diff --git a/third_party/i-slint-renderer-skia-1.17.1/Cargo.toml b/third_party/i-slint-renderer-skia-1.17.1/Cargo.toml new file mode 100644 index 0000000..63a1409 --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/Cargo.toml @@ -0,0 +1,339 @@ +# THIS FILE IS AUTOMATICALLY GENERATED BY CARGO +# +# When uploading crates to the registry Cargo will automatically +# "normalize" Cargo.toml files for maximal compatibility +# with all versions of Cargo and also rewrite `path` dependencies +# to registry (e.g., crates.io) dependencies. +# +# If you are reading this file be aware that the original Cargo.toml +# will likely look very different (and much more reasonable). +# See Cargo.toml.orig for the original contents. + +[package] +edition = "2024" +rust-version = "1.92" +name = "i-slint-renderer-skia" +version = "1.17.1" +authors = ["Slint Developers "] +build = "build.rs" +autolib = false +autobins = false +autoexamples = false +autotests = false +autobenches = false +description = "Skia based renderer for Slint" +homepage = "https://slint.dev" +readme = "README.md" +license = "GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0" +repository = "https://github.com/slint-ui/slint" + +[package.metadata.docs.rs] +rustdoc-args = ["--generate-link-to-definition"] + +[features] +default = ["softbuffer"] +kms = ["softbuffer/kms"] +opengl = [] +unstable-wgpu-28 = [ + "i-slint-core/unstable-wgpu-28", + "wgpu-28", +] +unstable-wgpu-29 = [ + "i-slint-core/unstable-wgpu-29", + "wgpu-29", +] +vulkan = [ + "skia-safe/vulkan", + "dep:ash", + "vulkano", +] +wayland = [ + "glutin/wayland", + "softbuffer/wayland", + "softbuffer/wayland-dlopen", +] +wgpu-28 = [ + "i-slint-core/wgpu-28", + "dep:wgpu-28", + "dep:spin_on", + "dep:foreign-types", + "dep:ash", + "dep:windows-core", +] +wgpu-29 = [ + "i-slint-core/wgpu-29", + "dep:wgpu-29", + "dep:spin_on", + "dep:ash", + "dep:windows-core", +] +x11 = [ + "glutin/x11", + "glutin/glx", + "softbuffer/x11", + "softbuffer/x11-dlopen", +] + +[lib] +name = "i_slint_renderer_skia" +path = "lib.rs" + +[dependencies.ash] +version = "^0.38.0" +optional = true + +[dependencies.cfg-if] +version = "1" + +[dependencies.clru] +version = "0.6.0" + +[dependencies.const-field-offset] +version = "0.2" + +[dependencies.derive_more] +version = "2.0.0" +features = [ + "deref", + "deref_mut", + "into", + "from", + "add", + "add_assign", + "mul", + "not", + "display", +] +default-features = false + +[dependencies.glow] +version = "0.17" + +[dependencies.i-slint-common] +version = "=1.17.1" +features = ["default"] +default-features = false + +[dependencies.i-slint-core] +version = "=1.17.1" +features = [ + "default", + "box-shadow-cache", + "shared-parley", +] +default-features = false + +[dependencies.i-slint-core-macros] +version = "=1.17.1" +features = ["default"] +default-features = false + +[dependencies.lyon_path] +version = "1.0" +default-features = false + +[dependencies.pin-weak] +version = "1" + +[dependencies.raw-window-handle] +version = "0.6" +features = ["std"] + +[dependencies.scoped-tls-hkt] +version = "0.1" + +[dependencies.skia-safe] +version = "0.99.0" +features = ["gl"] + +[dependencies.spin_on] +version = "0.1" +optional = true + +[dependencies.unicode-segmentation] +version = "1.12.0" + +[dependencies.vtable] +version = "0.4" + +[dependencies.vulkano] +version = "0.35.0" +optional = true +default-features = false + +[dependencies.wgpu-28] +version = "28" +optional = true +default-features = false +package = "wgpu" + +[dependencies.wgpu-29] +version = "29.0.4" +optional = true +default-features = false +package = "wgpu" + +[build-dependencies.cfg_aliases] +version = "0.2.0" + +[target.aarch64-apple-ios-sim.dependencies.objc2] +version = "0.6.0" +features = ["disable-encoding-assertions"] + +[target.'cfg(all(any(target_os = "ios", target_os="macos", target_os="windows", target_os="android", target_os="linux"), not(target_arch = "arm")))'.dependencies.skia-safe] +version = "0.99.0" +features = ["textlayout"] + +[target.'cfg(any(not(target_vendor = "apple"), target_os = "macos"))'.dependencies.glutin] +version = "0.32.0" +features = [ + "egl", + "wgl", +] +default-features = false + +[target.'cfg(not(any(target_vendor = "apple", target_family = "windows")))'.dependencies.skia-safe] +version = "0.99.0" +features = [ + "gl", + "vulkan", +] + +[target.'cfg(not(any(target_vendor = "apple", target_family = "windows")))'.dependencies.wgpu-28] +version = "28" +features = ["vulkan"] +optional = true +default-features = false +package = "wgpu" + +[target.'cfg(not(any(target_vendor = "apple", target_family = "windows")))'.dependencies.wgpu-29] +version = "29.0.4" +features = ["vulkan"] +optional = true +default-features = false +package = "wgpu" + +[target.'cfg(not(target_os = "android"))'.dependencies.bytemuck] +version = "1.13.1" + +[target.'cfg(not(target_os = "android"))'.dependencies.softbuffer] +version = "0.4.4" +optional = true +default-features = false + +[target.'cfg(target_family = "windows")'.dependencies.skia-safe] +version = "0.99.0" +features = ["d3d"] + +[target.'cfg(target_family = "windows")'.dependencies.wgpu-28] +version = "28" +features = ["dx12"] +optional = true +default-features = false +package = "wgpu" + +[target.'cfg(target_family = "windows")'.dependencies.wgpu-29] +version = "29.0.4" +features = ["dx12"] +optional = true +default-features = false +package = "wgpu" + +[target.'cfg(target_family = "windows")'.dependencies.windows] +version = "0.62" +features = [ + "Win32", + "Win32_System_Com", + "Win32_Graphics", + "Win32_Graphics_Dxgi", + "Win32_Graphics_Direct3D12", + "Win32_Graphics_Direct3D", + "Win32_Foundation", + "Win32_Graphics_Dxgi_Common", + "Win32_System_Threading", + "Win32_Security", +] + +[target.'cfg(target_family = "windows")'.dependencies.windows-core] +version = "0.62.0" +optional = true + +[target.'cfg(target_vendor = "apple")'.dependencies.foreign-types] +version = "0.5.0" +optional = true + +[target.'cfg(target_vendor = "apple")'.dependencies.objc2] +version = "0.6.0" + +[target.'cfg(target_vendor = "apple")'.dependencies.objc2-app-kit] +version = "0.3.2" +features = [ + "std", + "NSResponder", + "NSView", +] +default-features = false + +[target.'cfg(target_vendor = "apple")'.dependencies.objc2-core-foundation] +version = "0.3.2" +features = ["CFCGTypes"] +default-features = false + +[target.'cfg(target_vendor = "apple")'.dependencies.objc2-foundation] +version = "0.3.2" +features = [ + "std", + "NSGeometry", +] +default-features = false + +[target.'cfg(target_vendor = "apple")'.dependencies.objc2-metal] +version = "0.3.2" +features = [ + "std", + "MTLCommandQueue", + "MTLCommandBuffer", + "MTLDevice", + "MTLResource", + "MTLTexture", + "MTLTypes", +] +default-features = false + +[target.'cfg(target_vendor = "apple")'.dependencies.objc2-quartz-core] +version = "0.3.2" +features = [ + "std", + "objc2-metal", + "CALayer", + "CAMetalLayer", + "objc2-core-foundation", +] +default-features = false + +[target.'cfg(target_vendor = "apple")'.dependencies.raw-window-metal] +version = "1.0" + +[target.'cfg(target_vendor = "apple")'.dependencies.read-fonts] +version = "0.39" + +[target.'cfg(target_vendor = "apple")'.dependencies.skia-safe] +version = "0.99.0" +features = ["metal"] + +[target.'cfg(target_vendor = "apple")'.dependencies.wgpu-28] +version = "28" +features = ["metal"] +optional = true +default-features = false +package = "wgpu" + +[target.'cfg(target_vendor = "apple")'.dependencies.wgpu-29] +version = "29.0.4" +features = ["metal"] +optional = true +default-features = false +package = "wgpu" + +[target.'cfg(target_vendor = "apple")'.dependencies.write-fonts] +version = "0.48" diff --git a/third_party/i-slint-renderer-skia-1.17.1/Cargo.toml.orig b/third_party/i-slint-renderer-skia-1.17.1/Cargo.toml.orig new file mode 100644 index 0000000..2cf77b7 --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/Cargo.toml.orig @@ -0,0 +1,112 @@ +# Copyright © SixtyFPS GmbH +# SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +# cSpell: ignore CFCG +[package] +name = "i-slint-renderer-skia" +description = "Skia based renderer for Slint" +authors.workspace = true +edition = "2024" +homepage.workspace = true +license.workspace = true +repository.workspace = true +rust-version.workspace = true +version.workspace = true +build = "build.rs" + +[lib] +path = "lib.rs" + +# Note, these features need to be kept in sync (along with their defaults) in +# the C++ crate's CMakeLists.txt +[features] +wayland = ["glutin/wayland", "softbuffer/wayland", "softbuffer/wayland-dlopen"] +x11 = ["glutin/x11", "glutin/glx", "softbuffer/x11", "softbuffer/x11-dlopen"] +opengl = [] +vulkan = ["skia-safe/vulkan", "dep:ash", "vulkano"] +kms = ["softbuffer/kms"] +# wgpu-{28,29} enables the wgpu surface module for internal use (e.g. linuxkms DRM rendering). +# unstable-wgpu-{28,29} additionally exposes public API integration (GraphicsAPI, texture import). +wgpu-28 = ["i-slint-core/wgpu-28", "dep:wgpu-28", "dep:spin_on", "dep:foreign-types", "dep:ash", "dep:windows-core"] +unstable-wgpu-28 = ["i-slint-core/unstable-wgpu-28", "wgpu-28"] +wgpu-29 = ["i-slint-core/wgpu-29", "dep:wgpu-29", "dep:spin_on", "dep:ash", "dep:windows-core"] +unstable-wgpu-29 = ["i-slint-core/unstable-wgpu-29", "wgpu-29"] +default = ["softbuffer"] + +[dependencies] +i-slint-core = { workspace = true, features = ["default", "box-shadow-cache", "shared-parley"] } +i-slint-core-macros = { workspace = true, features = ["default"] } +i-slint-common = { workspace = true, features = ["default"] } + +const-field-offset = { version = "0.2", path = "../../../helper_crates/const-field-offset" } +vtable = { workspace = true } + +cfg-if = "1" +derive_more = { workspace = true } +lyon_path = { workspace = true } +pin-weak = "1" +scoped-tls-hkt = "0.1" +raw-window-handle = { version = "0.6", features = ["std"] } +clru = { workspace = true } + +skia-safe = { version = "0.99.0", features = ["gl"] } +glow = { workspace = true } +unicode-segmentation = { workspace = true } + +ash = { version = "^0.38.0", optional = true } +vulkano = { version = "0.35.0", optional = true, default-features = false } + +wgpu-28 = { workspace = true, optional = true } +wgpu-29 = { workspace = true, optional = true } +spin_on = { version = "0.1", optional = true } + +[target.'cfg(any(not(target_vendor = "apple"), target_os = "macos"))'.dependencies] +glutin = { workspace = true, default-features = false, features = ["egl", "wgl"] } + +[target.'cfg(not(target_os = "android"))'.dependencies] +# software renderer fallback +softbuffer = { workspace = true, default-features = false, optional = true } +bytemuck = { workspace = true } + +[target.'cfg(target_family = "windows")'.dependencies] +windows = { workspace = true, features = ["Win32", "Win32_System_Com", "Win32_Graphics", "Win32_Graphics_Dxgi", "Win32_Graphics_Direct3D12", "Win32_Graphics_Direct3D", "Win32_Foundation", "Win32_Graphics_Dxgi_Common", "Win32_System_Threading", "Win32_Security"] } +windows-core = { workspace = true, optional = true } +skia-safe = { version = "0.99.0", features = ["d3d"] } +wgpu-28 = { workspace = true, optional = true, features = ["dx12"] } +wgpu-29 = { workspace = true, optional = true, features = ["dx12"] } + +[target.'cfg(target_vendor = "apple")'.dependencies] +objc2 = { version = "0.6.0" } +objc2-metal = { version = "0.3.2", default-features = false, features = ["std", "MTLCommandQueue", "MTLCommandBuffer", "MTLDevice", "MTLResource", "MTLTexture", "MTLTypes"] } +objc2-foundation = { version = "0.3.2", default-features = false, features = ["std", "NSGeometry"] } +objc2-quartz-core = { version = "0.3.2", default-features = false, features = ["std", "objc2-metal", "CALayer", "CAMetalLayer", "objc2-core-foundation"] } +objc2-app-kit = { version = "0.3.2", default-features = false, features = ["std", "NSResponder", "NSView"] } +objc2-core-foundation = { version = "0.3.2", default-features = false, features = ["CFCGTypes"] } +skia-safe = { version = "0.99.0", features = ["metal"] } +raw-window-metal = "1.0" + +foreign-types = { version = "0.5.0", optional = true } +wgpu-28 = { workspace = true, optional = true, features = ["metal"] } +wgpu-29 = { workspace = true, optional = true, features = ["metal"] } + +read-fonts = { workspace = true } +# Pinned to 0.48 to align with read-fonts from parley and avoid duplicated dependencies +write-fonts = { version = "0.48" } + +[target.'cfg(not(any(target_vendor = "apple", target_family = "windows")))'.dependencies] +skia-safe = { version = "0.99.0", features = ["gl", "vulkan"] } +wgpu-28 = { workspace = true, optional = true, features = ["vulkan"] } +wgpu-29 = { workspace = true, optional = true, features = ["vulkan"] } + +[target.'cfg(all(any(target_os = "ios", target_os="macos", target_os="windows", target_os="android", target_os="linux"), not(target_arch = "arm")))'.dependencies] +skia-safe = { version = "0.99.0", features = ["textlayout"] } + +[target.aarch64-apple-ios-sim.dependencies] +# Disabling encoding assertions until https://github.com/madsmtm/objc2/issues/795 is fixed. +objc2 = { version = "0.6.0", features = ["disable-encoding-assertions"] } + +[build-dependencies] +cfg_aliases = { workspace = true } + +[package.metadata.docs.rs] +rustdoc-args = ["--generate-link-to-definition"] diff --git a/third_party/i-slint-renderer-skia-1.17.1/LICENSES/GPL-3.0-only.txt b/third_party/i-slint-renderer-skia-1.17.1/LICENSES/GPL-3.0-only.txt new file mode 100644 index 0000000..d41c0bd --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/LICENSES/GPL-3.0-only.txt @@ -0,0 +1,232 @@ +GNU GENERAL PUBLIC LICENSE +Version 3, 29 June 2007 + +Copyright © 2007 Free Software Foundation, Inc. + +Everyone is permitted to copy and distribute verbatim copies of this license document, but changing it is not allowed. + +Preamble + +The GNU General Public License is a free, copyleft license for software and other kinds of works. + +The licenses for most software and other practical works are designed to take away your freedom to share and change the works. By contrast, the GNU General Public License is intended to guarantee your freedom to share and change all versions of a program--to make sure it remains free software for all its users. We, the Free Software Foundation, use the GNU General Public License for most of our software; it applies also to any other work released this way by its authors. You can apply it to your programs, too. + +When we speak of free software, we are referring to freedom, not price. Our General Public Licenses are designed to make sure that you have the freedom to distribute copies of free software (and charge for them if you wish), that you receive source code or can get it if you want it, that you can change the software or use pieces of it in new free programs, and that you know you can do these things. + +To protect your rights, we need to prevent others from denying you these rights or asking you to surrender the rights. Therefore, you have certain responsibilities if you distribute copies of the software, or if you modify it: responsibilities to respect the freedom of others. + +For example, if you distribute copies of such a program, whether gratis or for a fee, you must pass on to the recipients the same freedoms that you received. You must make sure that they, too, receive or can get the source code. And you must show them these terms so they know their rights. + +Developers that use the GNU GPL protect your rights with two steps: (1) assert copyright on the software, and (2) offer you this License giving you legal permission to copy, distribute and/or modify it. + +For the developers' and authors' protection, the GPL clearly explains that there is no warranty for this free software. For both users' and authors' sake, the GPL requires that modified versions be marked as changed, so that their problems will not be attributed erroneously to authors of previous versions. + +Some devices are designed to deny users access to install or run modified versions of the software inside them, although the manufacturer can do so. This is fundamentally incompatible with the aim of protecting users' freedom to change the software. The systematic pattern of such abuse occurs in the area of products for individuals to use, which is precisely where it is most unacceptable. Therefore, we have designed this version of the GPL to prohibit the practice for those products. If such problems arise substantially in other domains, we stand ready to extend this provision to those domains in future versions of the GPL, as needed to protect the freedom of users. + +Finally, every program is threatened constantly by software patents. States should not allow patents to restrict development and use of software on general-purpose computers, but in those that do, we wish to avoid the special danger that patents applied to a free program could make it effectively proprietary. To prevent this, the GPL assures that patents cannot be used to render the program non-free. + +The precise terms and conditions for copying, distribution and modification follow. + +TERMS AND CONDITIONS + +0. Definitions. + +“This License” refers to version 3 of the GNU General Public License. + +“Copyright” also means copyright-like laws that apply to other kinds of works, such as semiconductor masks. + +“The Program” refers to any copyrightable work licensed under this License. Each licensee is addressed as “you”. “Licensees” and “recipients” may be individuals or organizations. + +To “modify” a work means to copy from or adapt all or part of the work in a fashion requiring copyright permission, other than the making of an exact copy. The resulting work is called a “modified version” of the earlier work or a work “based on” the earlier work. + +A “covered work” means either the unmodified Program or a work based on the Program. + +To “propagate” a work means to do anything with it that, without permission, would make you directly or secondarily liable for infringement under applicable copyright law, except executing it on a computer or modifying a private copy. Propagation includes copying, distribution (with or without modification), making available to the public, and in some countries other activities as well. + +To “convey” a work means any kind of propagation that enables other parties to make or receive copies. Mere interaction with a user through a computer network, with no transfer of a copy, is not conveying. + +An interactive user interface displays “Appropriate Legal Notices” to the extent that it includes a convenient and prominently visible feature that (1) displays an appropriate copyright notice, and (2) tells the user that there is no warranty for the work (except to the extent that warranties are provided), that licensees may convey the work under this License, and how to view a copy of this License. If the interface presents a list of user commands or options, such as a menu, a prominent item in the list meets this criterion. + +1. Source Code. +The “source code” for a work means the preferred form of the work for making modifications to it. “Object code” means any non-source form of a work. + +A “Standard Interface” means an interface that either is an official standard defined by a recognized standards body, or, in the case of interfaces specified for a particular programming language, one that is widely used among developers working in that language. + +The “System Libraries” of an executable work include anything, other than the work as a whole, that (a) is included in the normal form of packaging a Major Component, but which is not part of that Major Component, and (b) serves only to enable use of the work with that Major Component, or to implement a Standard Interface for which an implementation is available to the public in source code form. A “Major Component”, in this context, means a major essential component (kernel, window system, and so on) of the specific operating system (if any) on which the executable work runs, or a compiler used to produce the work, or an object code interpreter used to run it. + +The “Corresponding Source” for a work in object code form means all the source code needed to generate, install, and (for an executable work) run the object code and to modify the work, including scripts to control those activities. However, it does not include the work's System Libraries, or general-purpose tools or generally available free programs which are used unmodified in performing those activities but which are not part of the work. For example, Corresponding Source includes interface definition files associated with source files for the work, and the source code for shared libraries and dynamically linked subprograms that the work is specifically designed to require, such as by intimate data communication or control flow between those subprograms and other parts of the work. + +The Corresponding Source need not include anything that users can regenerate automatically from other parts of the Corresponding Source. + +The Corresponding Source for a work in source code form is that same work. + +2. Basic Permissions. +All rights granted under this License are granted for the term of copyright on the Program, and are irrevocable provided the stated conditions are met. This License explicitly affirms your unlimited permission to run the unmodified Program. The output from running a covered work is covered by this License only if the output, given its content, constitutes a covered work. This License acknowledges your rights of fair use or other equivalent, as provided by copyright law. + +You may make, run and propagate covered works that you do not convey, without conditions so long as your license otherwise remains in force. You may convey covered works to others for the sole purpose of having them make modifications exclusively for you, or provide you with facilities for running those works, provided that you comply with the terms of this License in conveying all material for which you do not control copyright. Those thus making or running the covered works for you must do so exclusively on your behalf, under your direction and control, on terms that prohibit them from making any copies of your copyrighted material outside their relationship with you. + +Conveying under any other circumstances is permitted solely under the conditions stated below. Sublicensing is not allowed; section 10 makes it unnecessary. + +3. Protecting Users' Legal Rights From Anti-Circumvention Law. +No covered work shall be deemed part of an effective technological measure under any applicable law fulfilling obligations under article 11 of the WIPO copyright treaty adopted on 20 December 1996, or similar laws prohibiting or restricting circumvention of such measures. + +When you convey a covered work, you waive any legal power to forbid circumvention of technological measures to the extent such circumvention is effected by exercising rights under this License with respect to the covered work, and you disclaim any intention to limit operation or modification of the work as a means of enforcing, against the work's users, your or third parties' legal rights to forbid circumvention of technological measures. + +4. Conveying Verbatim Copies. +You may convey verbatim copies of the Program's source code as you receive it, in any medium, provided that you conspicuously and appropriately publish on each copy an appropriate copyright notice; keep intact all notices stating that this License and any non-permissive terms added in accord with section 7 apply to the code; keep intact all notices of the absence of any warranty; and give all recipients a copy of this License along with the Program. + +You may charge any price or no price for each copy that you convey, and you may offer support or warranty protection for a fee. + +5. Conveying Modified Source Versions. +You may convey a work based on the Program, or the modifications to produce it from the Program, in the form of source code under the terms of section 4, provided that you also meet all of these conditions: + + a) The work must carry prominent notices stating that you modified it, and giving a relevant date. + + b) The work must carry prominent notices stating that it is released under this License and any conditions added under section 7. This requirement modifies the requirement in section 4 to “keep intact all notices”. + + c) You must license the entire work, as a whole, under this License to anyone who comes into possession of a copy. This License will therefore apply, along with any applicable section 7 additional terms, to the whole of the work, and all its parts, regardless of how they are packaged. This License gives no permission to license the work in any other way, but it does not invalidate such permission if you have separately received it. + + d) If the work has interactive user interfaces, each must display Appropriate Legal Notices; however, if the Program has interactive interfaces that do not display Appropriate Legal Notices, your work need not make them do so. + +A compilation of a covered work with other separate and independent works, which are not by their nature extensions of the covered work, and which are not combined with it such as to form a larger program, in or on a volume of a storage or distribution medium, is called an “aggregate” if the compilation and its resulting copyright are not used to limit the access or legal rights of the compilation's users beyond what the individual works permit. Inclusion of a covered work in an aggregate does not cause this License to apply to the other parts of the aggregate. + +6. Conveying Non-Source Forms. +You may convey a covered work in object code form under the terms of sections 4 and 5, provided that you also convey the machine-readable Corresponding Source under the terms of this License, in one of these ways: + + a) Convey the object code in, or embodied in, a physical product (including a physical distribution medium), accompanied by the Corresponding Source fixed on a durable physical medium customarily used for software interchange. + + b) Convey the object code in, or embodied in, a physical product (including a physical distribution medium), accompanied by a written offer, valid for at least three years and valid for as long as you offer spare parts or customer support for that product model, to give anyone who possesses the object code either (1) a copy of the Corresponding Source for all the software in the product that is covered by this License, on a durable physical medium customarily used for software interchange, for a price no more than your reasonable cost of physically performing this conveying of source, or (2) access to copy the Corresponding Source from a network server at no charge. + + c) Convey individual copies of the object code with a copy of the written offer to provide the Corresponding Source. This alternative is allowed only occasionally and noncommercially, and only if you received the object code with such an offer, in accord with subsection 6b. + + d) Convey the object code by offering access from a designated place (gratis or for a charge), and offer equivalent access to the Corresponding Source in the same way through the same place at no further charge. You need not require recipients to copy the Corresponding Source along with the object code. If the place to copy the object code is a network server, the Corresponding Source may be on a different server (operated by you or a third party) that supports equivalent copying facilities, provided you maintain clear directions next to the object code saying where to find the Corresponding Source. Regardless of what server hosts the Corresponding Source, you remain obligated to ensure that it is available for as long as needed to satisfy these requirements. + + e) Convey the object code using peer-to-peer transmission, provided you inform other peers where the object code and Corresponding Source of the work are being offered to the general public at no charge under subsection 6d. + +A separable portion of the object code, whose source code is excluded from the Corresponding Source as a System Library, need not be included in conveying the object code work. + +A “User Product” is either (1) a “consumer product”, which means any tangible personal property which is normally used for personal, family, or household purposes, or (2) anything designed or sold for incorporation into a dwelling. In determining whether a product is a consumer product, doubtful cases shall be resolved in favor of coverage. For a particular product received by a particular user, “normally used” refers to a typical or common use of that class of product, regardless of the status of the particular user or of the way in which the particular user actually uses, or expects or is expected to use, the product. A product is a consumer product regardless of whether the product has substantial commercial, industrial or non-consumer uses, unless such uses represent the only significant mode of use of the product. + +“Installation Information” for a User Product means any methods, procedures, authorization keys, or other information required to install and execute modified versions of a covered work in that User Product from a modified version of its Corresponding Source. The information must suffice to ensure that the continued functioning of the modified object code is in no case prevented or interfered with solely because modification has been made. + +If you convey an object code work under this section in, or with, or specifically for use in, a User Product, and the conveying occurs as part of a transaction in which the right of possession and use of the User Product is transferred to the recipient in perpetuity or for a fixed term (regardless of how the transaction is characterized), the Corresponding Source conveyed under this section must be accompanied by the Installation Information. But this requirement does not apply if neither you nor any third party retains the ability to install modified object code on the User Product (for example, the work has been installed in ROM). + +The requirement to provide Installation Information does not include a requirement to continue to provide support service, warranty, or updates for a work that has been modified or installed by the recipient, or for the User Product in which it has been modified or installed. Access to a network may be denied when the modification itself materially and adversely affects the operation of the network or violates the rules and protocols for communication across the network. + +Corresponding Source conveyed, and Installation Information provided, in accord with this section must be in a format that is publicly documented (and with an implementation available to the public in source code form), and must require no special password or key for unpacking, reading or copying. + +7. Additional Terms. +“Additional permissions” are terms that supplement the terms of this License by making exceptions from one or more of its conditions. Additional permissions that are applicable to the entire Program shall be treated as though they were included in this License, to the extent that they are valid under applicable law. If additional permissions apply only to part of the Program, that part may be used separately under those permissions, but the entire Program remains governed by this License without regard to the additional permissions. + +When you convey a copy of a covered work, you may at your option remove any additional permissions from that copy, or from any part of it. (Additional permissions may be written to require their own removal in certain cases when you modify the work.) You may place additional permissions on material, added by you to a covered work, for which you have or can give appropriate copyright permission. + +Notwithstanding any other provision of this License, for material you add to a covered work, you may (if authorized by the copyright holders of that material) supplement the terms of this License with terms: + + a) Disclaiming warranty or limiting liability differently from the terms of sections 15 and 16 of this License; or + + b) Requiring preservation of specified reasonable legal notices or author attributions in that material or in the Appropriate Legal Notices displayed by works containing it; or + + c) Prohibiting misrepresentation of the origin of that material, or requiring that modified versions of such material be marked in reasonable ways as different from the original version; or + + d) Limiting the use for publicity purposes of names of licensors or authors of the material; or + + e) Declining to grant rights under trademark law for use of some trade names, trademarks, or service marks; or + + f) Requiring indemnification of licensors and authors of that material by anyone who conveys the material (or modified versions of it) with contractual assumptions of liability to the recipient, for any liability that these contractual assumptions directly impose on those licensors and authors. + +All other non-permissive additional terms are considered “further restrictions” within the meaning of section 10. If the Program as you received it, or any part of it, contains a notice stating that it is governed by this License along with a term that is a further restriction, you may remove that term. If a license document contains a further restriction but permits relicensing or conveying under this License, you may add to a covered work material governed by the terms of that license document, provided that the further restriction does not survive such relicensing or conveying. + +If you add terms to a covered work in accord with this section, you must place, in the relevant source files, a statement of the additional terms that apply to those files, or a notice indicating where to find the applicable terms. + +Additional terms, permissive or non-permissive, may be stated in the form of a separately written license, or stated as exceptions; the above requirements apply either way. + +8. Termination. +You may not propagate or modify a covered work except as expressly provided under this License. Any attempt otherwise to propagate or modify it is void, and will automatically terminate your rights under this License (including any patent licenses granted under the third paragraph of section 11). + +However, if you cease all violation of this License, then your license from a particular copyright holder is reinstated (a) provisionally, unless and until the copyright holder explicitly and finally terminates your license, and (b) permanently, if the copyright holder fails to notify you of the violation by some reasonable means prior to 60 days after the cessation. + +Moreover, your license from a particular copyright holder is reinstated permanently if the copyright holder notifies you of the violation by some reasonable means, this is the first time you have received notice of violation of this License (for any work) from that copyright holder, and you cure the violation prior to 30 days after your receipt of the notice. + +Termination of your rights under this section does not terminate the licenses of parties who have received copies or rights from you under this License. If your rights have been terminated and not permanently reinstated, you do not qualify to receive new licenses for the same material under section 10. + +9. Acceptance Not Required for Having Copies. +You are not required to accept this License in order to receive or run a copy of the Program. Ancillary propagation of a covered work occurring solely as a consequence of using peer-to-peer transmission to receive a copy likewise does not require acceptance. However, nothing other than this License grants you permission to propagate or modify any covered work. These actions infringe copyright if you do not accept this License. Therefore, by modifying or propagating a covered work, you indicate your acceptance of this License to do so. + +10. Automatic Licensing of Downstream Recipients. +Each time you convey a covered work, the recipient automatically receives a license from the original licensors, to run, modify and propagate that work, subject to this License. You are not responsible for enforcing compliance by third parties with this License. + +An “entity transaction” is a transaction transferring control of an organization, or substantially all assets of one, or subdividing an organization, or merging organizations. If propagation of a covered work results from an entity transaction, each party to that transaction who receives a copy of the work also receives whatever licenses to the work the party's predecessor in interest had or could give under the previous paragraph, plus a right to possession of the Corresponding Source of the work from the predecessor in interest, if the predecessor has it or can get it with reasonable efforts. + +You may not impose any further restrictions on the exercise of the rights granted or affirmed under this License. For example, you may not impose a license fee, royalty, or other charge for exercise of rights granted under this License, and you may not initiate litigation (including a cross-claim or counterclaim in a lawsuit) alleging that any patent claim is infringed by making, using, selling, offering for sale, or importing the Program or any portion of it. + +11. Patents. +A “contributor” is a copyright holder who authorizes use under this License of the Program or a work on which the Program is based. The work thus licensed is called the contributor's “contributor version”. + +A contributor's “essential patent claims” are all patent claims owned or controlled by the contributor, whether already acquired or hereafter acquired, that would be infringed by some manner, permitted by this License, of making, using, or selling its contributor version, but do not include claims that would be infringed only as a consequence of further modification of the contributor version. For purposes of this definition, “control” includes the right to grant patent sublicenses in a manner consistent with the requirements of this License. + +Each contributor grants you a non-exclusive, worldwide, royalty-free patent license under the contributor's essential patent claims, to make, use, sell, offer for sale, import and otherwise run, modify and propagate the contents of its contributor version. + +In the following three paragraphs, a “patent license” is any express agreement or commitment, however denominated, not to enforce a patent (such as an express permission to practice a patent or covenant not to sue for patent infringement). To “grant” such a patent license to a party means to make such an agreement or commitment not to enforce a patent against the party. + +If you convey a covered work, knowingly relying on a patent license, and the Corresponding Source of the work is not available for anyone to copy, free of charge and under the terms of this License, through a publicly available network server or other readily accessible means, then you must either (1) cause the Corresponding Source to be so available, or (2) arrange to deprive yourself of the benefit of the patent license for this particular work, or (3) arrange, in a manner consistent with the requirements of this License, to extend the patent license to downstream recipients. “Knowingly relying” means you have actual knowledge that, but for the patent license, your conveying the covered work in a country, or your recipient's use of the covered work in a country, would infringe one or more identifiable patents in that country that you have reason to believe are valid. + +If, pursuant to or in connection with a single transaction or arrangement, you convey, or propagate by procuring conveyance of, a covered work, and grant a patent license to some of the parties receiving the covered work authorizing them to use, propagate, modify or convey a specific copy of the covered work, then the patent license you grant is automatically extended to all recipients of the covered work and works based on it. + +A patent license is “discriminatory” if it does not include within the scope of its coverage, prohibits the exercise of, or is conditioned on the non-exercise of one or more of the rights that are specifically granted under this License. You may not convey a covered work if you are a party to an arrangement with a third party that is in the business of distributing software, under which you make payment to the third party based on the extent of your activity of conveying the work, and under which the third party grants, to any of the parties who would receive the covered work from you, a discriminatory patent license (a) in connection with copies of the covered work conveyed by you (or copies made from those copies), or (b) primarily for and in connection with specific products or compilations that contain the covered work, unless you entered into that arrangement, or that patent license was granted, prior to 28 March 2007. + +Nothing in this License shall be construed as excluding or limiting any implied license or other defenses to infringement that may otherwise be available to you under applicable patent law. + +12. No Surrender of Others' Freedom. +If conditions are imposed on you (whether by court order, agreement or otherwise) that contradict the conditions of this License, they do not excuse you from the conditions of this License. If you cannot convey a covered work so as to satisfy simultaneously your obligations under this License and any other pertinent obligations, then as a consequence you may not convey it at all. For example, if you agree to terms that obligate you to collect a royalty for further conveying from those to whom you convey the Program, the only way you could satisfy both those terms and this License would be to refrain entirely from conveying the Program. + +13. Use with the GNU Affero General Public License. +Notwithstanding any other provision of this License, you have permission to link or combine any covered work with a work licensed under version 3 of the GNU Affero General Public License into a single combined work, and to convey the resulting work. The terms of this License will continue to apply to the part which is the covered work, but the special requirements of the GNU Affero General Public License, section 13, concerning interaction through a network will apply to the combination as such. + +14. Revised Versions of this License. +The Free Software Foundation may publish revised and/or new versions of the GNU General Public License from time to time. Such new versions will be similar in spirit to the present version, but may differ in detail to address new problems or concerns. + +Each version is given a distinguishing version number. If the Program specifies that a certain numbered version of the GNU General Public License “or any later version” applies to it, you have the option of following the terms and conditions either of that numbered version or of any later version published by the Free Software Foundation. If the Program does not specify a version number of the GNU General Public License, you may choose any version ever published by the Free Software Foundation. + +If the Program specifies that a proxy can decide which future versions of the GNU General Public License can be used, that proxy's public statement of acceptance of a version permanently authorizes you to choose that version for the Program. + +Later license versions may give you additional or different permissions. However, no additional obligations are imposed on any author or copyright holder as a result of your choosing to follow a later version. + +15. Disclaimer of Warranty. +THERE IS NO WARRANTY FOR THE PROGRAM, TO THE EXTENT PERMITTED BY APPLICABLE LAW. EXCEPT WHEN OTHERWISE STATED IN WRITING THE COPYRIGHT HOLDERS AND/OR OTHER PARTIES PROVIDE THE PROGRAM “AS IS” WITHOUT WARRANTY OF ANY KIND, EITHER EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE. THE ENTIRE RISK AS TO THE QUALITY AND PERFORMANCE OF THE PROGRAM IS WITH YOU. SHOULD THE PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF ALL NECESSARY SERVICING, REPAIR OR CORRECTION. + +16. Limitation of Liability. +IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MODIFIES AND/OR CONVEYS THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING OUT OF THE USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED TO LOSS OF DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY YOU OR THIRD PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER PROGRAMS), EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE POSSIBILITY OF SUCH DAMAGES. + +17. Interpretation of Sections 15 and 16. +If the disclaimer of warranty and limitation of liability provided above cannot be given local legal effect according to their terms, reviewing courts shall apply local law that most closely approximates an absolute waiver of all civil liability in connection with the Program, unless a warranty or assumption of liability accompanies a copy of the Program in return for a fee. + +END OF TERMS AND CONDITIONS + +How to Apply These Terms to Your New Programs + +If you develop a new program, and you want it to be of the greatest possible use to the public, the best way to achieve this is to make it free software which everyone can redistribute and change under these terms. + +To do so, attach the following notices to the program. It is safest to attach them to the start of each source file to most effectively state the exclusion of warranty; and each file should have at least the “copyright” line and a pointer to where the full notice is found. + + + Copyright (C) + + This program is free software: you can redistribute it and/or modify it under the terms of the GNU General Public License as published by the Free Software Foundation, either version 3 of the License, or (at your option) any later version. + + This program is distributed in the hope that it will be useful, but WITHOUT ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for more details. + + You should have received a copy of the GNU General Public License along with this program. If not, see . + +Also add information on how to contact you by electronic and paper mail. + +If the program does terminal interaction, make it output a short notice like this when it starts in an interactive mode: + + Copyright (C) + This program comes with ABSOLUTELY NO WARRANTY; for details type `show w'. + This is free software, and you are welcome to redistribute it under certain conditions; type `show c' for details. + +The hypothetical commands `show w' and `show c' should show the appropriate parts of the General Public License. Of course, your program's commands might be different; for a GUI interface, you would use an “about box”. + +You should also get your employer (if you work as a programmer) or school, if any, to sign a “copyright disclaimer” for the program, if necessary. For more information on this, and how to apply and follow the GNU GPL, see . + +The GNU General Public License does not permit incorporating your program into proprietary programs. If your program is a subroutine library, you may consider it more useful to permit linking proprietary applications with the library. If this is what you want to do, use the GNU Lesser General Public License instead of this License. But first, please read . diff --git a/third_party/i-slint-renderer-skia-1.17.1/LICENSES/LicenseRef-Slint-Royalty-free-2.0.md b/third_party/i-slint-renderer-skia-1.17.1/LICENSES/LicenseRef-Slint-Royalty-free-2.0.md new file mode 100644 index 0000000..f087009 --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/LICENSES/LicenseRef-Slint-Royalty-free-2.0.md @@ -0,0 +1,43 @@ +# Slint Royalty-free Desktop, Mobile, and Web Applications License + +Version 2.0 + +## Preamble + +Slint is a toolkit that can be used to build user interfaces for applications. Slint (hereafter referred to as **Software**) is made available under different licenses by SixtyFPS GmbH incorporated at Oranienburger Str. 44, 16540 Hohen Neuendorf, Germany (**SixtyFPS**). The **Slint Royalty-free Desktop, Mobile, and Web Applications License** is suitable for those who develop desktop, mobile, or web applications and do not want to use open source components under copyleft licenses. + +## 1. Grant of Rights + +SixtyFPS hereby grants You a world-wide, royalty-free, non-exclusive license to use, reproduce, make available, modify, display, perform, distribute the Software as part of a Desktop, Mobile, or Web Application. + +A **Desktop Application** is a computer program that is designed to run on a general-purpose computer (PC or notebook), typically installed and executed locally on the computer's operating system. + +A **Mobile Application** is a computer program that is designed to run on a general-purpose mobile computer (mobile phone or tablet), typically installed and executed locally on the computer's operating system. + +A **Web Application** is a computer program that is designed to run in the sandbox environment provided by a web browser. + +Desktop Application, Mobile Application, and Web Application are hereafter referred to as **Application**. + +## 2. License Conditions - Attribution + +You may distribute the Software as part of an Application, modified or unmodified, provided that You do either of the following: + +(a) Display the [`AboutSlint`](https://docs.slint.dev/latest/docs/slint/reference/std-widgets/misc/aboutslint/) widget in an "About" screen or dialog that is accessible from the top level menu of the Application. In the absence of such a screen or dialog, display the widget in the "Splash Screen" of the Application. + +(b) Display the [Slint attribution badge](https://github.com/slint-ui/slint/tree/master/logo/MadeWithSlint-logo-whitebg.png) on a public webpage, preferably where the binaries of your Application can be downloaded from, in such a way that it can be easily found by any visitor to that page. + +## 3. Limitations + +The License does not permit to distribute or make the Software publicly available alone and without integration into an Application. For this purpose you may use the Software under the GNU General Public License, version 3. + +The License does not permit the use of the Software within Embedded Systems. An **Embedded System** is a computer system designed to perform a specific task within a larger mechanical or electrical system. + +The License does not permit the distribution of Application that exposes the APIs, in part or in total, of the Software. + +You may not remove or alter any license notices (including copyright notices, disclaimers of warranty, or limitations of liability) contained within the source code form of the Software. + +## 4. Warranty and Liability + +SixtyFPS is only liable for conflicting rights of third parties if SixtyFPS was aware of these rights without informing you. Unless required by applicable law or agreed to in writing, SixtyFPS provides the Software on an "as is" basis, without warranties or conditions of any kind, either express or implied, including, without limitation, any warranties or conditions of merchantability, or fitness for a particular purpose. + +Unless required by law, SixtyFPS won't be liable for any direct, indirect, incidental, or consequential damages arising in any way out of the use of the Software. diff --git a/third_party/i-slint-renderer-skia-1.17.1/LICENSES/LicenseRef-Slint-Software-3.0.md b/third_party/i-slint-renderer-skia-1.17.1/LICENSES/LicenseRef-Slint-Software-3.0.md new file mode 100644 index 0000000..e5db7d0 --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/LICENSES/LicenseRef-Slint-Software-3.0.md @@ -0,0 +1,117 @@ +# Slint Software License + +Version 3.0.5 + +## Preamble + +Slint is a toolkit that can be used to build user interfaces for applications. Slint (hereafter referred to as **Software**) is made available under different licenses by SixtyFPS GmbH incorporated at Oranienburger Str. 44, 16540 Hohen Neuendorf, Germany (**SixtyFPS**). The **Slint Software License** is suitable for those who do not want to use open source components under copyleft licenses. + +## 1. Grant of Rights + +SixtyFPS hereby grants You a world-wide, non-exclusive license to use, reproduce, make available, modify, display, perform, distribute the Software as part of a Desktop, Mobile, or Web Application or as part of an Embedded System (each of which is defined below). + +A **Desktop Application** is a computer program that is designed to run on a general-purpose computer (PC or notebook), typically installed and executed locally on the computer's operating system. + +A **Mobile Application** is a computer program that is designed to run on a general-purpose mobile computer (mobile phone or tablet), typically installed and executed locally on the computer's operating system. + +A **Web Application** is a computer program that is designed to run in the sandbox environment provided by a web browser. + +An **Embedded System** is a computer system designed to perform a specific task within a larger mechanical or electrical system. + +Desktop Application, Mobile Application, and Web Application are hereafter referred to as **Application**. + +## 2. License Conditions + +The grant of rights in section 1 are conditional, provided that You do all of the following: + +(a) You have purchased an appropriate **Paid License Plan** ([see Annex 1](#annex-1-paid-license-plans)) and the required amount of seats to cover all individual users of the Software associated with the designing, developing, or testing your Application or Embedded System. For clarity, each individual user is counted as one seat. + +(b) In the case that You are distributing the Software as part of an Embedded System, You have purchased an appropriate quantity of **Royalties**, one Royalty for each Embedded System. Royalties become due and payable upon manufacture of the Embedded System, regardless of whether such is subsequently sold, shipped, returned, replaced under warranty, or recalled. Payment of royalties is non-refundable under any circumstances. Royalties are not necessary for non-commercial projects, personal projects, and open source projects. + +## 3. Limitations + +The License does not permit to distribute or make the Software publicly available alone and without integration into an Application or into an Embedded System. For this purpose you may use the Software under the GNU General Public License, version 3. + +The License is limited to only the versions of Software that were made available to you under the Paid License Plan. For all other versions, you may use the Software under either the GNU General Public License, version 3 or the Slint Royalty-free Desktop, Mobile, and Web Applications License. + +The License does not permit the distribution of Application that exposes the APIs, in part or in total, of the Software. + +You may not remove or alter any license notices (including copyright notices, disclaimers of warranty, or limitations of liability) contained within the source code form of the Software. + +## 4. Audit Rights + +SixtyFPS or an independent certified auditor on SixtyFPS's behalf, may, upon its reasonable request, with 30 (thirty) days written notice, and at its sole expense, examine your books and records solely with respect to your use of the Software. Any such audit shall be conducted during regular business hours at your facilities and shall not unreasonably interfere with your business activities. The auditor shall not remove, copy, or redistribute any electronic material during an audit. If an audit reveals that you are using the Software in a way that is in material violation of the terms of this License, then you shall pay SixtyFPS reasonable costs of conducting the audit. The auditor shall only be allowed to report violations of the terms of this License, with a copy to you. You shall be provided the right to provide comments to the report before it is finalized. + +## 5. Termination + +(a) SixtyFPS may terminate this License if You materially breach any obligation hereunder, provided You have been provided notice of such breach and an opportunity to cure such breach during a period of not less than sixty (60) days following such notice. + +(b) You may terminate this License with or without cause upon no less than thirty (30) days advance written notice to SixtyFPS. + +(c) Upon termination of this License, You will immediately cease using, reproducing, making available, modifying, displaying, performing, distributing the Software and pay immediately any unpaid Fees and contractual penalties. + +(d) Sections 3 through 8 of this License will survive any termination of the License to the extent necessary to implement their objectives. + +## 6. Assignment + +You may assign this License, in whole or in part (whether by operation of law or otherwise), with prior consent from SixtyFPS, which shall not be unreasonably withheld or delayed. SixtyFPS may assign any of its rights or delegate any of its obligations hereunder with prior notice to You, provided that the successor maintains at least the same level of security, confidentiality, and data protection measures as in place at the time of assignment or delegation. Any attempt to assign this License other than in accordance with this Section 6 shall be null and void. + +## 7. Severability + +In the event that any provision of this License will, for any reason, be determined by any court of competent jurisdiction to be invalid, illegal or unenforceable in any respect, such invalidity, illegality or unenforceability will be interpreted as closely as possible so as not affect any other provision of this License, and such provision will further be modified by said court to permit its enforcement to the maximum extent permitted by law. + +## 8. Governing Law + +This Agreement shall be construed, interpreted, and governed by the laws of the Federal Republic of Germany. + +## Annex 1: Paid License Plans + +### Enterprise Plan + +The following is included as part of the plan + +(a) No restriction on the number of applications that are developed with Slint. + +(b) Live Preview. + +(c) Standard Support that includes addressing technical queries, troubleshooting, and rectifying bugs or errors (faults) present in the latest official stable release. + +(d) Perpetual Fallback License that allows continued use of a specific Slint version, including all bugfix updates (i.e., all Z releases within the X.Y.Z version), without an active subscription. This license applies only to those versions of Slint for which at least 12 consecutive months of subscription have been paid. + +(e) GUI Test Framework. + +### Small Enterprise Plan + +This plan is limited to individual companies with a staff headcount of upto 50 and either a turnover or balance sheet total of 10 million EUR or less. If You are a Small Enterprise, You are required to submit the self-assessment report generated from the EU SME Self-Assessment Tool (https://ec.europa.eu/info/funding-tenders/opportunities/portal/sme/public/organisation-name). + +The following is included as part of the plan + +(a) No restriction on the number of applications that are developed with Slint. + +(b) Live Preview. + +(c) Standard Support that includes addressing technical queries, troubleshooting, and rectifying bugs or errors (faults) present in the latest official stable release. + +The following can be purchased as an Add-On + +(a) Perpetual Fallback License that allows continued use of a specific Slint version, including all bugfix updates (i.e., all Z releases within the X.Y.Z version), without an active subscription. This license applies only to those versions of Slint for which at least 12 consecutive months of subscription have been paid. + +(b) GUI Test Framework. + +### Startup & Individual Plan + +This plan is limited to individuals and individual companies with a staff headcount of less than 10 and either a turnover or balance sheet total of 2 million EUR or less. If You are a Startup, you are required to submit the self-assessment report generated from the EU SME Self-Assessment Tool (https://ec.europa.eu/info/funding-tenders/opportunities/portal/sme/public/organisation-name). + +The following is included as part of the plan + +(a) No restriction on the number of applications that are developed with Slint. + +(b) Live Preview. + +The following can be purchased as an Add-On + +(a) Standard Support that includes addressing technical queries, troubleshooting, and rectifying bugs or errors (faults) present in the latest official stable release. + +(b) Perpetual Fallback License that allows continued use of a specific Slint version, including all bugfix updates (i.e., all Z releases within the X.Y.Z version), without an active subscription. This license applies only to those versions of Slint for which at least 12 consecutive months of subscription have been paid. + +(c) GUI Test Framework. diff --git a/third_party/i-slint-renderer-skia-1.17.1/README.md b/third_party/i-slint-renderer-skia-1.17.1/README.md new file mode 100644 index 0000000..fb83b36 --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/README.md @@ -0,0 +1,7 @@ + +**NOTE**: This library is an **internal** crate of the [Slint project](https://slint.dev). +This crate should **not be used directly** by applications using Slint. +You should use the `slint` crate instead. + +**WARNING**: This crate does not follow the semver convention for versioning and can +only be used with `version = "=x.y.z"` in Cargo.toml. diff --git a/third_party/i-slint-renderer-skia-1.17.1/build.rs b/third_party/i-slint-renderer-skia-1.17.1/build.rs new file mode 100644 index 0000000..e5c66df --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/build.rs @@ -0,0 +1,18 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +use cfg_aliases::cfg_aliases; + +fn main() { + // Setup cfg aliases + cfg_aliases! { + skia_backend_opengl: { any(feature = "opengl", not(any(target_vendor = "apple", target_family = "windows", target_arch = "wasm32"))) }, + skia_backend_metal: { all(target_vendor = "apple", not(feature = "opengl")) }, + skia_backend_vulkan: { feature = "vulkan" }, + skia_backend_software: { not(target_os = "android") }, + skia_backend_softbuffer: { all(skia_backend_software, feature = "softbuffer") }, + skia_windowed: { any(skia_backend_vulkan, skia_backend_opengl, skia_backend_metal, skia_backend_softbuffer) }, + } + + println!("cargo:rustc-check-cfg=cfg(slint_nightly_test)"); +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/cached_image.rs b/third_party/i-slint-renderer-skia-1.17.1/cached_image.rs new file mode 100644 index 0000000..58a39ef --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/cached_image.rs @@ -0,0 +1,151 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +use crate::PhysicalSize; +use i_slint_core::graphics::{ + Image, ImageCacheKey, ImageInner, IntRect, IntSize, OpaqueImage, OpaqueImageVTable, + SharedImageBuffer, cache as core_cache, +}; +use i_slint_core::items::ImageFit; +use i_slint_core::lengths::{LogicalSize, ScaleFactor}; + +struct SkiaCachedImage { + image: skia_safe::Image, + cache_key: ImageCacheKey, +} + +i_slint_core::OpaqueImageVTable_static! { + static SKIA_CACHED_IMAGE_VT for SkiaCachedImage +} + +impl OpaqueImage for SkiaCachedImage { + fn size(&self) -> IntSize { + IntSize::new(self.image.width() as u32, self.image.height() as u32) + } + + fn cache_key(&self) -> ImageCacheKey { + self.cache_key.clone() + } +} + +pub(crate) fn as_skia_image( + image: Image, + target_size_fn: &dyn Fn() -> LogicalSize, + image_fit: ImageFit, + scale_factor: ScaleFactor, + canvas: &skia_safe::Canvas, + surface: Option<&dyn crate::Surface>, +) -> Option { + let image_inner: &ImageInner = (&image).into(); + match image_inner { + ImageInner::None => None, + ImageInner::EmbeddedImage { buffer, cache_key } => { + let result = image_buffer_to_skia_image(buffer); + if let Some(img) = result.as_ref() { + core_cache::replace_cached_image( + cache_key.clone(), + ImageInner::BackendStorage(vtable::VRc::into_dyn(vtable::VRc::new( + SkiaCachedImage { image: img.clone(), cache_key: cache_key.clone() }, + ))), + ) + } + result + } + ImageInner::Svg(svg) => { + // Query target_width/height here again to ensure that changes will invalidate the item rendering cache. + let svg_size = svg.size(); + let fit = i_slint_core::graphics::fit( + image_fit, + target_size_fn() * scale_factor, + IntRect::from_size(svg_size.cast()), + scale_factor, + Default::default(), // We only care about the size, so alignments don't matter + Default::default(), + ); + let target_size = PhysicalSize::new( + svg_size.cast::().width * fit.source_to_target_x, + svg_size.cast::().height * fit.source_to_target_y, + ); + let pixels = match svg.render(Some(target_size.cast())).ok()? { + SharedImageBuffer::RGB8(_) => unreachable!(), + SharedImageBuffer::RGBA8(_) => unreachable!(), + SharedImageBuffer::RGBA8Premultiplied(pixels) => pixels, + }; + + let image_info = skia_safe::ImageInfo::new( + skia_safe::ISize::new(pixels.width() as i32, pixels.height() as i32), + skia_safe::ColorType::RGBA8888, + skia_safe::AlphaType::Premul, + None, + ); + + skia_safe::images::raster_from_data( + &image_info, + skia_safe::Data::new_copy(pixels.as_bytes()), + pixels.width() as usize * 4, + ) + } + ImageInner::StaticTextures(_) => todo!(), + ImageInner::BackendStorage(x) => { + vtable::VRc::borrow(x).downcast::().map(|x| x.image.clone()) + } + ImageInner::BorrowedOpenGLTexture(texture) => { + surface.and_then(|surface| surface.import_opengl_texture(canvas, texture)) + } + ImageInner::NineSlice(n) => as_skia_image( + n.image(), + target_size_fn, + ImageFit::Preserve, + scale_factor, + canvas, + surface, + ), + #[cfg(feature = "unstable-wgpu-29")] + ImageInner::WGPUTexture(any_wgpu_texture) => { + surface.and_then(|surface| surface.import_wgpu_texture(canvas, any_wgpu_texture)) + } + #[allow(unreachable_patterns)] + _ => None, + } +} + +fn image_buffer_to_skia_image(buffer: &SharedImageBuffer) -> Option { + let (data, bpl, size, color_type, alpha_type) = match buffer { + SharedImageBuffer::RGB8(pixels) => { + // RGB888 with one byte per component is not supported by Skia right now. Convert once to RGBA8 :-( + let rgba = pixels + .as_bytes() + .chunks(3) + .flat_map(|rgb| IntoIterator::into_iter([rgb[0], rgb[1], rgb[2], 255])) + .collect::>(); + ( + skia_safe::Data::new_copy(&rgba), + pixels.width() as usize * 4, + pixels.size(), + skia_safe::ColorType::RGBA8888, + skia_safe::AlphaType::Unpremul, + ) + } + SharedImageBuffer::RGBA8(pixels) => ( + skia_safe::Data::new_copy(pixels.as_bytes()), + pixels.width() as usize * 4, + pixels.size(), + skia_safe::ColorType::RGBA8888, + skia_safe::AlphaType::Unpremul, + ), + SharedImageBuffer::RGBA8Premultiplied(pixels) => ( + skia_safe::Data::new_copy(pixels.as_bytes()), + pixels.width() as usize * 4, + pixels.size(), + skia_safe::ColorType::RGBA8888, + skia_safe::AlphaType::Premul, + ), + }; + let image_info = skia_safe::ImageInfo::new( + skia_safe::ISize::new(size.width as i32, size.height as i32), + color_type, + alpha_type, + None, + ); + skia_safe::images::raster_from_data(&image_info, data, bpl) +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/d3d_surface.rs b/third_party/i-slint-renderer-skia-1.17.1/d3d_surface.rs new file mode 100644 index 0000000..5d7f3fe --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/d3d_surface.rs @@ -0,0 +1,446 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +// cSpell: ignore HRESULT +use i_slint_core::api::{PhysicalSize as PhysicalWindowSize, Window}; +use i_slint_core::graphics::RequestedGraphicsAPI; +use i_slint_core::partial_renderer::DirtyRegion; +use i_slint_core::platform::PlatformError; +use i_slint_core::renderer::DrawOutcome; +use std::cell::RefCell; +use std::sync::Arc; +use windows::Win32::Graphics::Direct3D::D3D_FEATURE_LEVEL_11_0; +use windows::Win32::Graphics::Dxgi::Common::DXGI_STANDARD_MULTISAMPLE_QUALITY_PATTERN; +use windows::core::Interface; + +use windows::Win32::Foundation::{DXGI_STATUS_OCCLUDED, HANDLE, HWND, S_OK}; +use windows::Win32::Graphics::Direct3D12::{ + D3D12_COMMAND_LIST_TYPE_DIRECT, D3D12_COMMAND_QUEUE_DESC, D3D12_FENCE_FLAG_NONE, + D3D12_RESOURCE_STATE_PRESENT, D3D12CreateDevice, ID3D12CommandQueue, ID3D12Device, ID3D12Fence, + ID3D12Resource, +}; +use windows::Win32::Graphics::Dxgi::{ + Common::{DXGI_FORMAT, DXGI_FORMAT_R8G8B8A8_UNORM, DXGI_SAMPLE_DESC}, + CreateDXGIFactory2, DXGI_ADAPTER_FLAG, DXGI_ADAPTER_FLAG_NONE, DXGI_ADAPTER_FLAG_SOFTWARE, + DXGI_CREATE_FACTORY_FLAGS, DXGI_PRESENT, DXGI_SWAP_CHAIN_DESC1, DXGI_SWAP_CHAIN_FLAG, + DXGI_SWAP_EFFECT_FLIP_DISCARD, DXGI_USAGE_RENDER_TARGET_OUTPUT, IDXGIFactory4, IDXGISwapChain3, +}; +use windows::Win32::System::Threading::{CreateEventW, INFINITE, WaitForSingleObjectEx}; + +use crate::SkiaSharedContext; + +trait MapToPlatformError { + fn map_platform_error(self, msg: &str) -> std::result::Result; +} + +impl MapToPlatformError for windows::core::Result { + fn map_platform_error(self, msg: &str) -> std::result::Result { + match self { + Ok(r) => Ok(r), + Err(hr) => Err(format!("{} failed. {:x}", msg, hr.code().0).into()), + } + } +} + +const DEFAULT_SURFACE_FORMAT: DXGI_FORMAT = DXGI_FORMAT_R8G8B8A8_UNORM; + +struct SwapChain { + command_queue: ID3D12CommandQueue, + swap_chain: IDXGISwapChain3, + surfaces: Option<[skia_safe::Surface; 2]>, + current_buffer_index: usize, + fence: ID3D12Fence, + fence_values: [u64; 2], + fence_event: HANDLE, + gr_context: skia_safe::gpu::DirectContext, +} + +impl SwapChain { + fn new( + command_queue: ID3D12CommandQueue, + device: &ID3D12Device, + mut gr_context: skia_safe::gpu::DirectContext, + window_handle: raw_window_handle::WindowHandle<'_>, + size: PhysicalWindowSize, + dxgi_factory: &IDXGIFactory4, + ) -> Result { + let swap_chain_desc = DXGI_SWAP_CHAIN_DESC1 { + Width: size.width, + Height: size.height, + Format: DEFAULT_SURFACE_FORMAT, + BufferCount: 2, + BufferUsage: DXGI_USAGE_RENDER_TARGET_OUTPUT, + SwapEffect: DXGI_SWAP_EFFECT_FLIP_DISCARD, + SampleDesc: DXGI_SAMPLE_DESC { Count: 1, ..Default::default() }, + ..Default::default() + }; + + let hwnd = match window_handle.as_raw() { + raw_window_handle::RawWindowHandle::Win32(raw_window_handle::Win32WindowHandle { + hwnd, + .. + }) => HWND(hwnd.get() as _), + _ => { + return Err( + format!("Metal surface is only supported with Win32WindowHandle").into() + ); + } + }; + + let swap_chain1 = unsafe { + dxgi_factory.CreateSwapChainForHwnd(&command_queue, hwnd, &swap_chain_desc, None, None) + } + .map_platform_error("unable to create D3D swap chain")?; + + let swap_chain: IDXGISwapChain3 = + swap_chain1.cast().map_platform_error("unable to cast swap chain 1 to v3")?; + + let fence = unsafe { device.CreateFence(0, D3D12_FENCE_FLAG_NONE) } + .map_platform_error("unable to create D3D12 fence")?; + + let fence_values = [0, 0]; + + let fence_event = unsafe { CreateEventW(None, false, false, None) } + .map_platform_error("error creating fence event")?; + + let current_buffer_index = unsafe { swap_chain.GetCurrentBackBufferIndex() } as usize; + + let surfaces = Some(Self::create_surfaces( + &swap_chain, + &mut gr_context, + size.width as _, + size.height as _, + )?); + + Ok(Self { + command_queue, + swap_chain, + surfaces, + current_buffer_index, + fence, + fence_event, + fence_values, + gr_context, + }) + } + + fn render_and_present( + &mut self, + callback: impl FnOnce(&mut skia_safe::Surface, &mut skia_safe::gpu::DirectContext, u8) -> T, + pre_present_callback: &RefCell>>, + ) -> Result { + let current_fence_value = self.fence_values[self.current_buffer_index]; + + self.current_buffer_index = unsafe { self.swap_chain.GetCurrentBackBufferIndex() } as usize; + self.wait_for_buffer(self.current_buffer_index)?; + + self.fence_values[self.current_buffer_index] = current_fence_value + 1; + + let surface = &mut (*self.surfaces.as_mut().unwrap())[self.current_buffer_index]; + + // TODO: pass correct buffer age + let result = callback(surface, &mut self.gr_context, 0); + + let info = Default::default(); + self.gr_context.flush_surface_with_access( + surface, + skia_safe::surface::BackendSurfaceAccess::Present, + &info, + ); + self.gr_context.submit(None); + + if let Some(pre_present_callback) = pre_present_callback.borrow_mut().as_mut() { + pre_present_callback(); + } + + let present_result = unsafe { self.swap_chain.Present(1, DXGI_PRESENT(0)) }; + if present_result != S_OK && present_result != DXGI_STATUS_OCCLUDED { + return Err(format!("Error presenting d3d swap chain: {:x}", present_result.0).into()); + } + + unsafe { + self.command_queue.Signal(&self.fence, self.fence_values[self.current_buffer_index]) + } + .map_platform_error("error setting up completion signal for d3d12 command queue")?; + + Ok(result) + } + + fn create_surfaces( + swap_chain: &IDXGISwapChain3, + gr_context: &mut skia_safe::gpu::DirectContext, + width: i32, + height: i32, + ) -> Result<[skia_safe::Surface; 2], PlatformError> { + let mut make_surface = |buffer_index| { + let buffer: ID3D12Resource = unsafe { swap_chain.GetBuffer(buffer_index) } + .map_err(|hr| format!("unable to retrieve swap chain back buffer: {hr}"))?; + + debug_assert_eq!(unsafe { buffer.GetDesc().Width }, width as u64); + debug_assert_eq!(unsafe { buffer.GetDesc().Height }, height as u32); + + let texture_info = skia_safe::gpu::d3d::TextureResourceInfo { + resource: buffer, + alloc: None, + resource_state: D3D12_RESOURCE_STATE_PRESENT, + format: DEFAULT_SURFACE_FORMAT, + sample_count: 1, + level_count: 1, + sample_quality_pattern: DXGI_STANDARD_MULTISAMPLE_QUALITY_PATTERN, + protected: skia_safe::gpu::Protected::No, + }; + let backend_texture = + skia_safe::gpu::backend_render_targets::make_d3d((width, height), &texture_info); + + skia_safe::gpu::surfaces::wrap_backend_render_target( + gr_context, + &backend_texture, + skia_safe::gpu::SurfaceOrigin::TopLeft, + skia_safe::ColorType::RGBA8888, + None, + None, + ) + .ok_or_else(|| format!("unable to create d3d skia backend render target")) + }; + + Ok([make_surface(0)?, make_surface(1)?]) + } + + fn resize( + &mut self, + width: u32, + height: u32, + ) -> Result<(), i_slint_core::platform::PlatformError> { + self.gr_context.flush_submit_and_sync_cpu(); + + self.wait_for_buffer(0)?; + self.wait_for_buffer(1)?; + + drop(self.surfaces.take()); + + unsafe { + self.swap_chain.ResizeBuffers( + 0, + width, + height, + DEFAULT_SURFACE_FORMAT, + DXGI_SWAP_CHAIN_FLAG(0), + ) + } + .map_platform_error("Error resizing swap chain buffers")?; + + self.surfaces = Some(Self::create_surfaces( + &self.swap_chain, + &mut self.gr_context, + width as i32, + height as i32, + )?); + Ok(()) + } + + fn wait_for_buffer(&mut self, buffer_index: usize) -> Result<(), PlatformError> { + if unsafe { self.fence.GetCompletedValue() } < self.fence_values[buffer_index] { + unsafe { + self.fence.SetEventOnCompletion(self.fence_values[buffer_index], self.fence_event) + } + .map_platform_error("error setting event on command queue completion")?; + + unsafe { + WaitForSingleObjectEx(self.fence_event, INFINITE, false); + } + } + Ok(()) + } +} + +/// This surface renders into the given window using Direct 3D. The provided display +/// argument is ignored, as it has no meaning on Windows. +pub struct D3DSurface { + swap_chain: RefCell, +} + +impl super::Surface for D3DSurface { + fn new( + _shared_context: &SkiaSharedContext, + window_handle: Arc, + _display_handle: Arc, + size: PhysicalWindowSize, + requested_graphics_api: Option, + ) -> Result { + if requested_graphics_api + .map_or(false, |api| !matches!(api, RequestedGraphicsAPI::Direct3D)) + { + return Err(format!("Requested non-Direct3D rendering with Direct3D renderer").into()); + } + + let factory_flags = 0; + /* + let factory_flags = dxgi1_3::DXGI_CREATE_FACTORY_DEBUG; + + { + let maybe_debug_interface: Result< + ComPtr, + HRESULT, + > = resolve_interface(|iid, ptr| unsafe { d3d12::D3D12GetDebugInterface(iid, ptr) }); + if let Ok(debug) = maybe_debug_interface { + unsafe { debug.EnableDebugLayer() }; + } + } + */ + + let dxgi_factory: IDXGIFactory4 = + unsafe { CreateDXGIFactory2(DXGI_CREATE_FACTORY_FLAGS(factory_flags)) } + .map_platform_error("unable to create DXGIFactory4")?; + + let mut software_adapter_index = None; + let use_warp = std::env::var("SLINT_D3D_USE_WARP").is_ok(); + + let adapter = { + let mut i = 0; + loop { + let adapter = match unsafe { dxgi_factory.EnumAdapters1(i) } { + Ok(adapter) => adapter, + Err(_) => break None, + }; + + let Ok(desc) = (unsafe { adapter.GetDesc1() }) else { + continue; + }; + + let adapter_is_warp = (DXGI_ADAPTER_FLAG(desc.Flags as i32) + & DXGI_ADAPTER_FLAG_SOFTWARE) + != DXGI_ADAPTER_FLAG_NONE; + + if adapter_is_warp { + if software_adapter_index.is_none() { + software_adapter_index = Some(i); + } + + if !use_warp { + i += 1; + // Select warp only if explicitly opted in via SLINT_D3D_USE_WARP + continue; + } + + // found warp adapter, requested warp? give it a try below + } else if use_warp { + // Don't select a non-warp adapter when warp is requested + i += 1; + continue; + } + + // Check to see whether the adapter supports Direct3D 12, but don't + // create the actual device yet. + if unsafe { + D3D12CreateDevice( + &adapter, + D3D_FEATURE_LEVEL_11_0, + std::ptr::null_mut::>(), + ) + } + .is_ok() + { + break Some(adapter); + } + + i += 1; + } + }; + + let adapter = adapter.map_or_else( + || { + let software_adapter_index = software_adapter_index + .ok_or_else(|| format!("unable to locate D3D software adapter"))?; + unsafe { dxgi_factory.EnumAdapters1(software_adapter_index) } + .map_err(|hr| format!("unable to create D3D software adapter: {hr}")) + }, + |adapter| Ok(adapter), + )?; + + let mut device: Option = None; + unsafe { D3D12CreateDevice(&adapter, D3D_FEATURE_LEVEL_11_0, &mut device) } + .map_platform_error("error calling D3D12CreateDevice")?; + let device = device.unwrap(); + + let queue: ID3D12CommandQueue = { + let desc = D3D12_COMMAND_QUEUE_DESC { + Type: D3D12_COMMAND_LIST_TYPE_DIRECT, + ..Default::default() + }; + + unsafe { device.CreateCommandQueue(&desc) } + .map_platform_error("Creating command queue")? + }; + + let backend_context = skia_safe::gpu::d3d::BackendContext { + adapter, + device: device.clone(), + queue: queue.clone(), + memory_allocator: None, + protected_context: skia_safe::gpu::Protected::No, + }; + + let gr_context = + unsafe { skia_safe::gpu::direct_contexts::make_d3d(&backend_context, None) } + .ok_or_else(|| format!("unable to create Skia D3D DirectContext"))?; + + let window_handle = window_handle + .window_handle() + .map_err(|e| format!("error obtaining window handle for skia d3d renderer: {e}"))?; + + let swap_chain = RefCell::new(SwapChain::new( + queue, + &device, + gr_context, + window_handle, + size, + &dxgi_factory, + )?); + + Ok(Self { swap_chain }) + } + + fn name(&self) -> &'static str { + "d3d" + } + + fn resize_event( + &self, + size: PhysicalWindowSize, + ) -> Result<(), i_slint_core::platform::PlatformError> { + self.swap_chain.borrow_mut().resize(size.width, size.height) + } + + fn render( + &self, + _window: &Window, + _size: PhysicalWindowSize, + callback: &dyn Fn( + &skia_safe::Canvas, + Option<&mut skia_safe::gpu::DirectContext>, + u8, + ) -> Option, + pre_present_callback: &RefCell>>, + ) -> Result { + self.swap_chain.borrow_mut().render_and_present( + |surface, gr_context, buffer_age| { + callback(surface.canvas(), Some(gr_context), buffer_age); + }, + pre_present_callback, + )?; + Ok(DrawOutcome::Success) + } + + fn bits_per_pixel(&self) -> Result { + let desc = unsafe { self.swap_chain.borrow().swap_chain.GetDesc() } + .map_platform_error("error getting swap chain description")?; + Ok(match desc.BufferDesc.Format { + DEFAULT_SURFACE_FORMAT => 32, + fmt @ _ => { + return Err( + format!("Skia D3D Renderer: Unsupported buffer format found {fmt:?}").into() + ); + } + }) + } +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/font_cache.rs b/third_party/i-slint-renderer-skia-1.17.1/font_cache.rs new file mode 100644 index 0000000..126e79e --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/font_cache.rs @@ -0,0 +1,106 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +use clru::CLruCache; +use i_slint_common::sharedfontique::HashedBlob; +use i_slint_core::textlayout::sharedparley::{fontique, parley}; +use std::cell::RefCell; +use std::collections::hash_map::DefaultHasher; +use std::hash::{Hash, Hasher}; +use std::num::NonZeroUsize; + +const FONT_CACHE_CAPACITY: NonZeroUsize = NonZeroUsize::new(64).unwrap(); + +pub struct FontCache { + font_mgr: skia_safe::FontMgr, + // Use HashedBlob in key to keep strong reference to font data blob, + // preventing eviction from fontique's shared cache (see commit 30a03cf). + // The u64 is a hash of variation settings (0 for base typefaces). + fonts: CLruCache<(HashedBlob, u32, u64), Option>, +} + +impl Default for FontCache { + fn default() -> Self { + Self { font_mgr: skia_safe::FontMgr::new(), fonts: CLruCache::new(FONT_CACHE_CAPACITY) } + } +} + +impl FontCache { + pub fn font_with_variations( + &mut self, + font: &parley::FontData, + synthesis: &fontique::Synthesis, + ) -> Option { + let variation_settings = synthesis.variation_settings(); + + let mut variations_hash = 0u64; + if !variation_settings.is_empty() { + let mut hasher = DefaultHasher::new(); + for &(tag, value) in variation_settings { + tag.to_be_bytes().hash(&mut hasher); + value.to_bits().hash(&mut hasher); + } + variations_hash = hasher.finish(); + } + + let key = (font.data.clone().into(), font.index, variations_hash); + + if let Some(cached) = self.fonts.get(&key) { + return cached.clone(); + } + + let mut typeface = self.load_typeface_internal(font); + + if !variation_settings.is_empty() { + typeface = typeface.and_then(|base| { + let coords: Vec = + variation_settings + .iter() + .map(|&(tag, value)| { + skia_safe::font_arguments::variation_position::Coordinate { + axis: skia_safe::FourByteTag::new(u32::from_be_bytes( + tag.to_be_bytes(), + )), + value, + } + }) + .collect(); + let position = + skia_safe::font_arguments::VariationPosition { coordinates: &coords }; + let args = skia_safe::FontArguments::new().set_variation_design_position(position); + base.clone_with_arguments(&args).or(Some(base)) + }); + } + + self.fonts.put(key, typeface.clone()); + typeface + } + + fn load_typeface_internal(&self, font: &parley::FontData) -> Option { + let typeface = self.font_mgr.new_from_data( + font.data.as_ref(), + if font.index > 0 { Some(font.index as _) } else { None }, + ); + + // Due to https://issues.skia.org/issues/310510989, fonts from true type collections + // with an index > 0 fail to load on macOS. As a workaround, we manually extract the font from the + // collection and load it as a single font. + #[cfg(target_vendor = "apple")] + if font.index > 0 + && typeface.is_none() + && let Some(typeface) = read_fonts::CollectionRef::new(font.data.as_ref()) + .ok() + .and_then(|ttc| ttc.get(font.index).ok()) + .map(|ttf| write_fonts::FontBuilder::new().copy_missing_tables(ttf).build()) + .and_then(|new_ttf| self.font_mgr.new_from_data(&new_ttf, None)) + { + return Some(typeface); + } + + typeface + } +} + +thread_local! { + pub static FONT_CACHE: RefCell = RefCell::new(Default::default()) +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/itemrenderer.rs b/third_party/i-slint-renderer-skia-1.17.1/itemrenderer.rs new file mode 100644 index 0000000..e8556ca --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/itemrenderer.rs @@ -0,0 +1,1264 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +// cSpell: ignore rrect skpath + +use std::pin::Pin; + +use super::{PhysicalBorderRadius, PhysicalLength, PhysicalPoint, PhysicalRect, PhysicalSize}; +use i_slint_core::graphics::ApproxEq; +use i_slint_core::graphics::boxshadowcache::BoxShadowCache; +use i_slint_core::graphics::euclid::num::Zero; +use i_slint_core::graphics::euclid::{self, Vector2D}; +use i_slint_core::item_rendering::{ + CachedRenderingData, ItemCache, ItemRenderer, ItemRendererFeatures, LayerRenderer, RenderImage, + RenderText, +}; +use i_slint_core::items::{ImageFit, ImageRendering, ItemRc, Layer, Opacity, RenderingResult}; +use i_slint_core::lengths::{ + LogicalBorderRadius, LogicalLength, LogicalPoint, LogicalPx, LogicalRect, LogicalSize, + LogicalVector, PhysicalPx, RectLengths, ScaleFactor, SizeLengths, logical_size_from_api, +}; +use i_slint_core::textlayout::sharedparley::{self, GlyphRenderer, fontique}; +use i_slint_core::window::WindowInner; +use i_slint_core::{Brush, Color, SharedString}; +use skia_safe::{Matrix, TileMode}; + +pub type SkiaBoxShadowCache = BoxShadowCache; + +#[derive(Clone, Copy)] +struct RenderState { + alpha: f32, + transform: i_slint_core::lengths::ItemTransform, +} + +pub struct SkiaItemRenderer<'a> { + pub canvas: &'a skia_safe::Canvas, + pub scale_factor: ScaleFactor, + pub window: &'a i_slint_core::api::Window, + surface: Option<&'a dyn crate::Surface>, + state_stack: Vec, + current_state: RenderState, + image_cache: &'a ItemCache>, + layer_cache: &'a ItemCache>, + path_cache: &'a ItemCache, skia_safe::Path)>>, + text_layout_cache: &'a sharedparley::TextLayoutCache, + box_shadow_cache: &'a mut SkiaBoxShadowCache, +} + +impl<'a> SkiaItemRenderer<'a> { + pub fn new( + canvas: &'a skia_safe::Canvas, + window: &'a i_slint_core::api::Window, + surface: Option<&'a dyn crate::Surface>, + image_cache: &'a ItemCache>, + layer_cache: &'a ItemCache>, + path_cache: &'a ItemCache, skia_safe::Path)>>, + text_layout_cache: &'a sharedparley::TextLayoutCache, + box_shadow_cache: &'a mut SkiaBoxShadowCache, + ) -> Self { + Self { + canvas, + scale_factor: ScaleFactor::new(window.scale_factor()), + window, + surface, + state_stack: Vec::new(), + current_state: RenderState { + alpha: 1.0, + transform: i_slint_core::lengths::ItemTransform::identity(), + }, + image_cache, + layer_cache, + path_cache, + text_layout_cache, + box_shadow_cache, + } + } + + fn default_paint(&self) -> Option { + if self.current_state.alpha.approx_eq(&1.0) { + None + } else { + let mut paint = skia_safe::Paint::default(); + paint.set_alpha_f(self.current_state.alpha); + Some(paint) + } + } + + fn render_drop_shadow_image( + canvas: &skia_safe::Canvas, + shadow_options: &i_slint_core::graphics::boxshadowcache::BoxShadowOptions, + ) -> Option { + let blur = shadow_options.blur.get(); + let spread = shadow_options.spread.get(); + + let shape_w = (shadow_options.width.get() + 2. * spread).max(0.); + let shape_h = (shadow_options.height.get() + 2. * spread).max(0.); + if shape_w <= 0. || shape_h <= 0. { + return None; + } + // CSS rule: outer corner radius after spread = max(0, r + spread). + let shape_radius = (shadow_options.radius + PhysicalBorderRadius::new_uniform(spread)) + .max(Default::default()); + + let canvas_size: skia_safe::Size = (shape_w + 2. * blur, shape_h + 2. * blur).into(); + + let image_info = skia_safe::ImageInfo::new( + canvas_size.to_ceil(), + skia_safe::ColorType::RGBA8888, + skia_safe::AlphaType::Premul, + None, + ); + + // The shape is centered in the canvas with `blur` padding on all sides so the Gaussian blur + // has room to fade out into transparency. + let rounded_rect = to_skia_rrect( + &PhysicalRect::new(PhysicalPoint::new(blur, blur), PhysicalSize::new(shape_w, shape_h)), + &shape_radius, + ); + + let mut paint = skia_safe::Paint::default(); + paint.set_color(to_skia_color(&shadow_options.color)); + paint.set_anti_alias(true); + if blur > 0. { + paint.set_mask_filter(skia_safe::MaskFilter::blur( + skia_safe::BlurStyle::Normal, + blur / 2., + None, + )); + } + + let mut surface = canvas.new_surface(&image_info, None)?; + let surface_canvas = surface.canvas(); + surface_canvas.clear(skia_safe::Color::TRANSPARENT); + surface_canvas.draw_rrect(rounded_rect, &paint); + Some(surface.image_snapshot()) + } + + fn render_inset_shadow_image( + canvas: &skia_safe::Canvas, + shadow_options: &i_slint_core::graphics::boxshadowcache::BoxShadowOptions, + ) -> Option { + let width = shadow_options.width.get(); + let height = shadow_options.height.get(); + if width < 1. || height < 1. { + return None; + } + let blur = shadow_options.blur.get(); + let spread = shadow_options.spread.get(); + let radius = shadow_options.radius; + let offset_x = shadow_options.offset_x_inset; + let offset_y = shadow_options.offset_y_inset; + + // Image is sized to the rectangle's geometry; the geometry rrect serves as the clip so the + // outer blurred edge stays hidden. + let canvas_size = skia_safe::ISize::new(width.ceil() as i32, height.ceil() as i32); + let image_info = skia_safe::ImageInfo::new( + canvas_size, + skia_safe::ColorType::RGBA8888, + skia_safe::AlphaType::Premul, + None, + ); + + let geometry_rrect = to_skia_rrect( + &PhysicalRect::new(PhysicalPoint::zero(), PhysicalSize::new(width, height)), + &radius, + ); + + // Inner "hole" rrect: geometry inset by spread on each side, translated by offset. + // CSS: inner radius = max(0, radius - spread). + let inner_rect = skia_safe::Rect::new( + spread + offset_x, + spread + offset_y, + width - spread + offset_x, + height - spread + offset_y, + ); + let inner_radius = + (radius - PhysicalBorderRadius::new_uniform(spread)).max(Default::default()); + let inner_rrect = to_skia_rrect( + &PhysicalRect::new( + PhysicalPoint::new(inner_rect.left, inner_rect.top), + PhysicalSize::new(inner_rect.width(), inner_rect.height()), + ), + &inner_radius, + ); + + // Outer rect inflated well beyond the geometry so its blurred edge falls outside the clip. + let inflate = blur + spread.abs() + offset_x.abs() + offset_y.abs() + 16.; + let outer_rect = + skia_safe::Rect::new(-inflate, -inflate, width + inflate, height + inflate); + + let mut path_builder = skia_safe::PathBuilder::new(); + path_builder.set_fill_type(skia_safe::PathFillType::EvenOdd); + path_builder.add_rect(outer_rect, None, None); + path_builder.add_rrect(inner_rrect, None, None); + let path = path_builder.detach(); + + let mut paint = skia_safe::Paint::default(); + paint.set_color(to_skia_color(&shadow_options.color)); + paint.set_anti_alias(true); + if blur > 0. { + paint.set_mask_filter(skia_safe::MaskFilter::blur( + skia_safe::BlurStyle::Normal, + blur / 2., + None, + )); + } + + let mut surface = canvas.new_surface(&image_info, None)?; + let surface_canvas = surface.canvas(); + surface_canvas.clear(skia_safe::Color::TRANSPARENT); + surface_canvas.clip_rrect(geometry_rrect, None, true); + surface_canvas.draw_path(&path, &paint); + Some(surface.image_snapshot()) + } + + fn brush_to_paint( + &self, + brush: Brush, + width: PhysicalLength, + height: PhysicalLength, + ) -> Option { + let (mut paint, shader) = Self::brush_to_shader( + self.default_paint().unwrap_or_default(), + brush, + width, + height, + self.scale_factor.get(), + )?; + paint.set_shader(Some(shader)); + + Some(paint) + } + + fn brush_to_shader( + mut paint: skia_safe::Paint, + brush: Brush, + width: PhysicalLength, + height: PhysicalLength, + scale_factor: f32, + ) -> Option<(skia_safe::Paint, skia_safe::Shader)> { + if brush.is_transparent() { + return None; + } + + match brush { + Brush::SolidColor(color) => Some(skia_safe::shaders::color(to_skia_color(&color))), + + Brush::LinearGradient(g) => { + let (start, end) = i_slint_core::graphics::line_for_angle( + g.angle(), + [width.get(), height.get()].into(), + ); + let (colors, pos): (Vec, Vec<_>) = g + .stops() + .map(|s| (skia_safe::Color4f::from(to_skia_color(&s.color)), s.position)) + .unzip(); + + paint.set_dither(true); + + let gradient_colors = + skia_safe::gradient::Colors::new(&colors, Some(&*pos), TileMode::Clamp, None); + let gradient = skia_safe::gradient::Gradient::new( + gradient_colors, + skia_safe::gradient::Interpolation { + in_premul: skia_safe::gradient::interpolation::InPremul::Yes, + ..Default::default() + }, + ); + skia_safe::gradient::shaders::linear_gradient( + (skia_safe::Point::new(start.x, start.y), skia_safe::Point::new(end.x, end.y)), + &gradient, + None, + ) + } + Brush::RadialGradient(g) => { + let (colors, pos): (Vec, Vec<_>) = g + .stops() + .map(|s| (skia_safe::Color4f::from(to_skia_color(&s.color)), s.position)) + .unzip(); + let (cx, cy) = g.center_or_default_scaled(width.get(), height.get(), scale_factor); + let circle_scale = + g.radius_or_default_scaled(width.get(), height.get(), scale_factor); + + paint.set_dither(true); + + let gradient_colors = + skia_safe::gradient::Colors::new(&colors, Some(&*pos), TileMode::Clamp, None); + let gradient = skia_safe::gradient::Gradient::new( + gradient_colors, + skia_safe::gradient::Interpolation { + in_premul: skia_safe::gradient::interpolation::InPremul::Yes, + ..Default::default() + }, + ); + let mut local_matrix = skia_safe::Matrix::scale((circle_scale, circle_scale)); + local_matrix.post_translate((cx, cy)); + skia_safe::gradient::shaders::radial_gradient( + (skia_safe::Point::new(0., 0.), 1.), + &gradient, + &local_matrix, + ) + } + Brush::ConicGradient(g) => { + let (colors, pos): (Vec, Vec<_>) = g + .stops() + .map(|s| (skia_safe::Color4f::from(to_skia_color(&s.color)), s.position)) + .unzip(); + let (cx, cy) = g.center_or_default_scaled(width.get(), height.get(), scale_factor); + + paint.set_dither(true); + + // Skia's sweep gradient uses 0 degrees at 3 o'clock (east) + // We want 0 degrees at 12 o'clock (north), so we need to rotate by -90 degrees + let center = skia_safe::Point::new(cx, cy); + let gradient_colors = + skia_safe::gradient::Colors::new(&colors, Some(&*pos), TileMode::Clamp, None); + let gradient = skia_safe::gradient::Gradient::new( + gradient_colors, + skia_safe::gradient::Interpolation::default(), + ); + skia_safe::gradient::shaders::sweep_gradient( + center, + (0.0, 360.0), + &gradient, + &skia_safe::Matrix::rotate_deg_pivot(-90.0, center), + ) + } + _ => None, + } + .map(|shader| (paint, shader)) + } + + fn colorize_image( + &mut self, + image: skia_safe::Image, + colorize_brush: Brush, + ) -> Option { + let image_info = skia_safe::ImageInfo::new( + image.dimensions(), + skia_safe::ColorType::RGBA8888, + skia_safe::AlphaType::Premul, + None, + ); + + Self::brush_to_shader( + skia_safe::Paint::default(), // Don't use the renderer's default paint because alpha is applied later + colorize_brush, + PhysicalLength::new(image.width() as f32), + PhysicalLength::new(image.height() as f32), + self.scale_factor.get(), + ) + .map(|(mut paint, colorize_shader)| { + let mut surface = self.canvas.new_surface(&image_info, None)?; + let canvas = surface.canvas(); + canvas.clear(skia_safe::Color::TRANSPARENT); + + paint.set_image_filter(skia_safe::image_filters::blend( + skia_safe::BlendMode::SrcIn, + skia_safe::image_filters::image(image, None, None, None), + skia_safe::image_filters::shader(colorize_shader, None), + None, + )); + canvas.draw_paint(&paint); + Some(surface.image_snapshot()) + })? + } + + fn draw_image_impl( + &mut self, + item_rc: &ItemRc, + item: Pin<&dyn RenderImage>, + dest_rect: PhysicalRect, + ) { + let tiling = item.tiling(); + + // TODO: avoid doing creating an SkImage multiple times when the same source is used in multiple image elements + let skia_image = self.image_cache.get_or_update_cache_entry(item_rc, || { + let image = item.source(); + super::cached_image::as_skia_image( + image, + &|| item.target_size(), + if tiling != Default::default() { ImageFit::Preserve } else { item.image_fit() }, + self.scale_factor, + self.canvas, + self.surface, + ) + .and_then(|skia_image| { + let brush = item.colorize(); + if !brush.is_transparent() { + self.colorize_image(skia_image, brush) + } else { + Some(skia_image) + } + }) + }); + + let Some(skia_image) = skia_image else { return }; + let source = item.source(); + let source_size = source.size(); + if source_size.is_empty() { + // Not sure how this can happen, but we've seen with #6280 + // that somehow we end up with a `skia_safe::Image` but a zero + // source size. + return; + } + let fits = if let i_slint_core::ImageInner::NineSlice(nine) = + <&i_slint_core::ImageInner>::from(&source) + { + i_slint_core::graphics::fit9slice( + source_size.cast(), + nine.1, + dest_rect.size, + self.scale_factor, + item.alignment(), + tiling, + ) + .collect::>() + } else { + vec![i_slint_core::graphics::fit( + item.image_fit(), + dest_rect.size, + item.source_clip().unwrap_or_else(|| euclid::Rect::from_size(source_size.cast())), + self.scale_factor, + item.alignment(), + tiling, + )] + }; + + let _saved_canvas = self.pixel_align_origin_auto_restore(); + for fit in fits { + self.canvas.save(); + + let dst = to_skia_rect(&PhysicalRect::new(fit.offset, fit.size)); + self.canvas.clip_rect(dst, None, None); + let src = skia_safe::IRect::from_xywh( + skia_image.width() * fit.clip_rect.origin.x / source_size.width as i32, + skia_image.height() * fit.clip_rect.origin.y / source_size.height as i32, + skia_image.width() * fit.clip_rect.size.width / source_size.width as i32, + skia_image.height() * fit.clip_rect.size.height / source_size.height as i32, + ); + + let filter_mode: skia_safe::sampling_options::SamplingOptions = + match item.rendering() { + ImageRendering::Pixelated => skia_safe::sampling_options::FilterMode::Nearest, + ImageRendering::Smooth | _ => skia_safe::sampling_options::FilterMode::Linear, + } + .into(); + + if let Some(tiled_offset) = fit.tiled { + let matrix = Matrix::translate(((fit.offset.x as i32), (fit.offset.y as i32))) + * Matrix::scale(( + fit.source_to_target_x * source_size.width as f32 + / skia_image.width() as f32, + fit.source_to_target_y * source_size.height as f32 + / skia_image.height() as f32, + )) + * Matrix::translate((-(tiled_offset.x as i32), -(tiled_offset.y as i32))); + if let Some(shader) = skia_image + .make_subset( + self.canvas + .recording_context() + .as_mut() + .map(|c| c.as_recorder() as &mut dyn skia_safe::Recorder), + src, + skia_safe::image::RequiredProperties::default(), + ) + .and_then(|i| { + i.to_shader((TileMode::Repeat, TileMode::Repeat), filter_mode, &matrix) + }) + { + let mut paint = self.default_paint().unwrap_or_default(); + paint.set_shader(shader); + self.canvas.draw_paint(&paint); + } + } else { + let transform = + skia_safe::Matrix::rect_2_rect(skia_safe::Rect::from(src), dst, None) + .unwrap_or_default(); + self.canvas.concat(&transform); + self.canvas.draw_image_with_sampling_options( + skia_image.clone(), + skia_safe::Point::default(), + filter_mode, + self.default_paint().as_ref(), + ); + } + + self.canvas.restore(); + } + } + + fn render_and_blend_layer(&mut self, item_rc: &ItemRc) -> RenderingResult { + if let Some((layer_offset, layer_image)) = + i_slint_core::item_rendering::render_layer(self, item_rc) + { + self.canvas.translate(skia_safe::Vector::from((layer_offset.x, layer_offset.y))); + let _saved_canvas = self.pixel_align_origin_auto_restore(); + self.canvas.draw_image_with_sampling_options( + layer_image, + skia_safe::Point::default(), + skia_safe::sampling_options::FilterMode::Linear, + self.default_paint().as_ref(), + ); + } + RenderingResult::ContinueRenderingWithoutChildren + } + + // Same as pixel_align_origin_auto_restore() but can be used across function calls where + // `&self` is needed. Returns true if the caller must call `restore()` on `self.canvas`. + fn save_canvas_and_pixel_align_origin(&self) -> bool { + let local_to_device = self.canvas.local_to_device_as_3x3(); + if !local_to_device.is_translate() || local_to_device.is_identity() { + return false; + } + let Some(device_to_local) = local_to_device.invert() else { + return false; + }; + let mut target_point = local_to_device.map_point(skia_safe::Point::default()); + + target_point.x = target_point.x.round(); + target_point.y = target_point.y.round(); + + self.canvas.save(); + + self.canvas.translate(device_to_local.map_point(target_point)); + + true + } + + fn pixel_align_origin_auto_restore(&self) -> Option> { + let local_to_device = self.canvas.local_to_device_as_3x3(); + if !local_to_device.is_translate() || local_to_device.is_identity() { + return None; + } + let device_to_local = local_to_device.invert()?; + let mut target_point = local_to_device.map_point(skia_safe::Point::default()); + + target_point.x = target_point.x.round(); + target_point.y = target_point.y.round(); + + let restore_point = skia_safe::AutoCanvasRestore::guard(self.canvas, true); + + self.canvas.translate(device_to_local.map_point(target_point)); + + Some(restore_point) + } +} + +impl ItemRenderer for SkiaItemRenderer<'_> { + fn draw_rectangle( + &mut self, + rect: Pin<&dyn i_slint_core::item_rendering::RenderRectangle>, + _self_rc: &i_slint_core::items::ItemRc, + size: LogicalSize, + _cache: &CachedRenderingData, + ) { + let geometry = PhysicalRect::from(size * self.scale_factor); + if geometry.is_empty() { + return; + } + + let paint = match self.brush_to_paint( + rect.background(), + geometry.width_length(), + geometry.height_length(), + ) { + Some(paint) => paint, + None => return, + }; + self.canvas.draw_rect(to_skia_rect(&geometry), &paint); + } + + fn draw_border_rectangle( + &mut self, + rect: Pin<&dyn i_slint_core::item_rendering::RenderBorderRectangle>, + _self_rc: &i_slint_core::items::ItemRc, + size: LogicalSize, + _: &CachedRenderingData, + ) { + let mut geometry = PhysicalRect::from(size * self.scale_factor); + if geometry.is_empty() { + return; + } + + // Save the original element bounds for gradient positioning. The CSS model positions + // gradients relative to the border box (full element), but adjust_rect_and_border_for_inner_drawing + // shrinks geometry before we create the paint, which would shift the gradient center inward. + let original_width = geometry.width_length(); + let original_height = geometry.height_length(); + + let border_color = rect.border_color(); + let opaque_border = border_color.is_opaque(); + let mut border_width = if border_color.is_transparent() { + PhysicalLength::new(0.) + } else { + rect.border_width() * self.scale_factor + }; + + // Radius of rounded rect if we were to just fill the rectangle, without a border. + let mut fill_radius = rect.border_radius() * self.scale_factor; + // Skia's border radius on stroke is in the middle of the border. But we want it to be the radius of the rectangle itself. + // This is incorrect if fill_radius < border_width/2, but this can't be fixed. Better to have a radius a bit too big than no radius at all + fill_radius = fill_radius.outer(border_width / 2. + PhysicalLength::new(0.01)); + let stroke_border_radius = fill_radius.inner(border_width / 2.); + + let (background_rect, border_rect) = if opaque_border { + // In CSS the border is entirely towards the inside of the boundary + // geometry, while in femtovg the line with for a stroke is 50% in- + // and 50% outwards. We choose the CSS model, so the inner rectangle + // is adjusted accordingly. + adjust_rect_and_border_for_inner_drawing(&mut geometry, &mut border_width); + + let rounded_rect = to_skia_rrect(&geometry, &stroke_border_radius); + + (rounded_rect, rounded_rect) + } else { + let background_rect = to_skia_rrect(&geometry, &fill_radius); + + // In CSS the border is entirely towards the inside of the boundary + // geometry, while in femtovg the line with for a stroke is 50% in- + // and 50% outwards. We choose the CSS model, so the inner rectangle + // is adjusted accordingly. + adjust_rect_and_border_for_inner_drawing(&mut geometry, &mut border_width); + + let border_rect = to_skia_rrect(&geometry, &stroke_border_radius); + + (background_rect, border_rect) + }; + + if let Some(mut fill_paint) = + self.brush_to_paint(rect.background(), original_width, original_height) + { + fill_paint.set_style(skia_safe::PaintStyle::Fill); + if !background_rect.is_rect() { + fill_paint.set_anti_alias(true); + } + self.canvas.draw_rrect(background_rect, &fill_paint); + } + + if border_width.get() > 0.0 + && let Some(mut border_paint) = + self.brush_to_paint(border_color, original_width, original_height) + { + border_paint.set_style(skia_safe::PaintStyle::Stroke); + border_paint.set_stroke_width(border_width.get()); + if !border_rect.is_rect() { + border_paint.set_anti_alias(true); + } + self.canvas.draw_rrect(border_rect, &border_paint); + } + } + + fn draw_window_background( + &mut self, + _rect: Pin<&dyn i_slint_core::item_rendering::RenderRectangle>, + _self_rc: &ItemRc, + _size: LogicalSize, + _cache: &CachedRenderingData, + ) { + // The background is drawn directly by FemtoVG renderer (via clear_color, if necessary). + } + + fn draw_image( + &mut self, + image: Pin<&dyn RenderImage>, + self_rc: &ItemRc, + size: LogicalSize, + _cache: &CachedRenderingData, + ) { + let geometry = PhysicalRect::from(size * self.scale_factor); + if geometry.is_empty() { + return; + } + self.draw_image_impl(self_rc, image, geometry); + } + + fn draw_text( + &mut self, + text: Pin<&dyn RenderText>, + self_rc: &i_slint_core::items::ItemRc, + size: LogicalSize, + _cache: &CachedRenderingData, + ) { + let restore = self.save_canvas_and_pixel_align_origin(); + sharedparley::draw_text(self, text, Some(self_rc), size, Some(self.text_layout_cache)); + if restore { + self.canvas.restore(); + } + } + + fn draw_text_input( + &mut self, + text_input: Pin<&i_slint_core::items::TextInput>, + self_rc: &i_slint_core::items::ItemRc, + size: LogicalSize, + ) { + let restore = self.save_canvas_and_pixel_align_origin(); + sharedparley::draw_text_input(self, text_input, self_rc, size, None); + if restore { + self.canvas.restore(); + } + } + + fn draw_path( + &mut self, + path: Pin<&i_slint_core::items::Path>, + item_rc: &i_slint_core::items::ItemRc, + size: LogicalSize, + ) { + let geometry = PhysicalRect::from(size * self.scale_factor); + + let (physical_offset, skpath): (crate::euclid::Vector2D, _) = + match self.path_cache.get_or_update_cache_entry(item_rc, || { + let (logical_offset, path_events): (crate::euclid::Vector2D, _) = + path.fitted_path_events(item_rc)?; + + let mut builder = skia_safe::PathBuilder::new(); + + for x in path_events.iter() { + match x { + lyon_path::Event::Begin { at } => { + builder.move_to(to_skia_point( + LogicalPoint::from_untyped(at) * self.scale_factor, + )); + } + lyon_path::Event::Line { from: _, to } => { + builder.line_to(to_skia_point( + LogicalPoint::from_untyped(to) * self.scale_factor, + )); + } + lyon_path::Event::Quadratic { from: _, ctrl, to } => { + builder.quad_to( + to_skia_point(LogicalPoint::from_untyped(ctrl) * self.scale_factor), + to_skia_point(LogicalPoint::from_untyped(to) * self.scale_factor), + ); + } + + lyon_path::Event::Cubic { from: _, ctrl1, ctrl2, to } => { + builder.cubic_to( + to_skia_point( + LogicalPoint::from_untyped(ctrl1) * self.scale_factor, + ), + to_skia_point( + LogicalPoint::from_untyped(ctrl2) * self.scale_factor, + ), + to_skia_point(LogicalPoint::from_untyped(to) * self.scale_factor), + ); + } + lyon_path::Event::End { last: _, first: _, close } => { + if close { + builder.close(); + } + } + } + } + + (logical_offset * self.scale_factor, builder.detach()).into() + }) { + Some(offset_and_path) => offset_and_path, + None => return, + }; + + self.canvas.translate((physical_offset.x, physical_offset.y)); + + let anti_alias = path.anti_alias(); + + // For Path elements with conic gradients, we need to handle the viewbox transformation + let viewbox_width = path.viewbox_width(); + let viewbox_height = path.viewbox_height(); + + let paint = if viewbox_width > 0.0 && viewbox_height > 0.0 { + // If there's a viewbox, we need to create the gradient in viewbox space + // and then transform it to the actual size + let scale_x = geometry.width() / viewbox_width; + let scale_y = geometry.height() / viewbox_height; + + let paint = self.default_paint().unwrap_or_default(); + if let Some((mut paint, shader)) = Self::brush_to_shader( + paint, + path.fill(), + PhysicalLength::new(viewbox_width), + PhysicalLength::new(viewbox_height), + 1.0, + ) { + // Apply the viewbox transformation to the shader + let transform = skia_safe::Matrix::scale((scale_x, scale_y)); + paint.set_shader(shader.with_local_matrix(&transform)); + Some(paint) + } else { + None + } + } else { + self.brush_to_paint(path.fill(), geometry.width_length(), geometry.height_length()) + }; + + if let Some(mut fill_paint) = paint { + fill_paint.set_anti_alias(anti_alias); + self.canvas.draw_path(&skpath, &fill_paint); + } + if let Some(mut border_paint) = + self.brush_to_paint(path.stroke(), geometry.width_length(), geometry.height_length()) + { + border_paint.set_anti_alias(anti_alias); + border_paint.set_stroke_width((path.stroke_width() * self.scale_factor).get()); + border_paint.set_stroke_cap(match path.stroke_line_cap() { + i_slint_core::items::LineCap::Round => skia_safe::PaintCap::Round, + i_slint_core::items::LineCap::Square => skia_safe::PaintCap::Square, + i_slint_core::items::LineCap::Butt | _ => skia_safe::PaintCap::Butt, + }); + border_paint.set_stroke_join(match path.stroke_line_join() { + i_slint_core::items::LineJoin::Round => skia_safe::PaintJoin::Round, + i_slint_core::items::LineJoin::Bevel => skia_safe::PaintJoin::Bevel, + i_slint_core::items::LineJoin::Miter | _ => skia_safe::PaintJoin::Miter, + }); + border_paint.set_stroke_miter(path.stroke_miter_limit()); + border_paint.set_stroke(true); + self.canvas.draw_path(&skpath, &border_paint); + } + } + + fn draw_box_shadow( + &mut self, + box_shadow: Pin<&i_slint_core::items::BoxShadow>, + self_rc: &i_slint_core::items::ItemRc, + _size: LogicalSize, + ) { + let offset = LogicalPoint::from_lengths(box_shadow.offset_x(), box_shadow.offset_y()) + * self.scale_factor; + let inset = box_shadow.inset(); + let spread = box_shadow.spread() * self.scale_factor; + + // Drop shadow with no offset / blur / spread is invisible. + if !inset + && offset.x == 0. + && offset.y == 0. + && box_shadow.blur() == LogicalLength::zero() + && spread == PhysicalLength::zero() + { + return; + } + + let cached_shadow_image = self.box_shadow_cache.get_box_shadow( + self_rc, + self.image_cache, + box_shadow, + self.scale_factor, + |shadow_options| { + if shadow_options.inset { + Self::render_inset_shadow_image(self.canvas, shadow_options) + } else { + Self::render_drop_shadow_image(self.canvas, shadow_options) + } + }, + ); + + let cached_shadow_image = match cached_shadow_image { + Some(img) => img, + None => return, + }; + + if inset { + // Inset image is sized exactly to the geometry; blit at origin. + self.canvas.draw_image( + cached_shadow_image, + skia_safe::Point::new(0., 0.), + self.default_paint().as_ref(), + ); + } else { + let blur = box_shadow.blur() * self.scale_factor; + let pad = blur.get() + spread.get().max(0.); + self.canvas.draw_image( + cached_shadow_image, + to_skia_point(offset - PhysicalPoint::new(pad, pad).to_vector()), + self.default_paint().as_ref(), + ); + } + } + + fn combine_clip( + &mut self, + rect: LogicalRect, + radius: LogicalBorderRadius, + border_width: LogicalLength, + ) -> bool { + let mut rect = rect * self.scale_factor; + let mut border_width = border_width * self.scale_factor; + // In CSS the border is entirely towards the inside of the boundary + // geometry, while in femtovg the line with for a stroke is 50% in- + // and 50% outwards. We choose the CSS model, so the inner rectangle + // is adjusted accordingly. + adjust_rect_and_border_for_inner_drawing(&mut rect, &mut border_width); + + let radius = radius * self.scale_factor; + let rounded_rect = to_skia_rrect(&rect, &radius); + self.canvas.clip_rrect(rounded_rect, None, true); + self.canvas.local_clip_bounds().is_some() + } + + fn get_current_clip(&self) -> LogicalRect { + from_skia_rect(&self.canvas.local_clip_bounds().unwrap_or_default()) / self.scale_factor + } + + fn translate(&mut self, distance: LogicalVector) { + self.current_state.transform = self.current_state.transform.pre_translate(distance.cast()); + let distance = distance * self.scale_factor; + self.canvas.translate(skia_safe::Vector::from((distance.x, distance.y))); + } + + fn current_transform(&self) -> i_slint_core::lengths::ItemTransform { + self.current_state.transform + } + + fn rotate(&mut self, angle_in_degrees: f32) { + self.current_state.transform = + self.current_state.transform.pre_rotate(euclid::Angle::degrees(angle_in_degrees)); + self.canvas.rotate(angle_in_degrees, None); + } + + fn scale(&mut self, x_factor: f32, y_factor: f32) { + self.current_state.transform = self.current_state.transform.pre_scale(x_factor, y_factor); + self.canvas.scale((x_factor, y_factor)); + } + + fn apply_opacity(&mut self, opacity: f32) { + self.current_state.alpha *= opacity; + } + + fn save_state(&mut self) { + self.canvas.save(); + self.state_stack.push(self.current_state); + } + + fn restore_state(&mut self) { + self.current_state = self.state_stack.pop().unwrap(); + self.canvas.restore(); + } + + fn scale_factor(&self) -> f32 { + self.scale_factor.get() + } + + fn draw_cached_pixmap( + &mut self, + item_rc: &i_slint_core::items::ItemRc, + update_fn: &dyn Fn(&mut dyn FnMut(u32, u32, &[u8])), + ) { + let skia_image = self.image_cache.get_or_update_cache_entry(item_rc, || { + let mut cached_image = None; + update_fn(&mut |width: u32, height: u32, data: &[u8]| { + let image_info = skia_safe::ImageInfo::new( + skia_safe::ISize::new(width as i32, height as i32), + skia_safe::ColorType::RGBA8888, + skia_safe::AlphaType::Premul, + None, + ); + cached_image = skia_safe::images::raster_from_data( + &image_info, + skia_safe::Data::new_copy(data), + width as usize * 4, + ); + }); + cached_image + }); + let skia_image = match skia_image { + Some(img) => img, + None => return, + }; + let _saved_canvas = self.pixel_align_origin_auto_restore(); + self.canvas.draw_image(skia_image, skia_safe::Point::default(), None); + } + + fn draw_string(&mut self, string: &str, color: i_slint_core::Color) { + sharedparley::draw_text( + self, + std::pin::pin!((SharedString::from(string), Brush::from(color))), + None, + logical_size_from_api(self.window.size().to_logical(self.scale_factor())), + None, + ); + } + + fn draw_image_direct(&mut self, image: i_slint_core::graphics::Image) { + let skia_image = super::cached_image::as_skia_image( + image.clone(), + &|| LogicalSize::from_untyped(image.size().cast()), + ImageFit::Fill, + self.scale_factor, + self.canvas, + self.surface, + ); + + let skia_image = match skia_image { + Some(img) => img, + None => return, + }; + + self.canvas.draw_image( + skia_image, + skia_safe::Point::default(), + self.default_paint().as_ref(), + ); + } + + fn window(&self) -> &i_slint_core::window::WindowInner { + i_slint_core::window::WindowInner::from_pub(self.window) + } + + fn as_any(&mut self) -> Option<&mut dyn core::any::Any> { + None + } + + fn visit_opacity( + &mut self, + opacity_item: Pin<&Opacity>, + item_rc: &ItemRc, + _size: LogicalSize, + ) -> RenderingResult { + let opacity = opacity_item.opacity(); + if Opacity::need_layer(item_rc, opacity) { + self.canvas.save_layer_alpha(None, (opacity * 255.) as u32); + self.state_stack.push(self.current_state); + self.current_state.alpha = 1.0; + + let window_adapter = WindowInner::from_pub(self.window).window_adapter(); + + i_slint_core::item_rendering::render_item_children( + self, + item_rc.item_tree(), + item_rc.index() as isize, + &window_adapter, + ); + + self.current_state = self.state_stack.pop().unwrap(); + self.canvas.restore(); + RenderingResult::ContinueRenderingWithoutChildren + } else { + self.apply_opacity(opacity); + RenderingResult::ContinueRenderingChildren + } + } + + fn visit_layer( + &mut self, + layer_item: Pin<&Layer>, + self_rc: &ItemRc, + _size: LogicalSize, + ) -> RenderingResult { + if layer_item.cache_rendering_hint() { + self.render_and_blend_layer(self_rc) + } else { + self.image_cache.release(self_rc); + RenderingResult::ContinueRenderingChildren + } + } +} + +impl<'a> LayerRenderer<'a> for SkiaItemRenderer<'a> { + type LayerTarget = skia_safe::Surface; + type Image = skia_safe::Image; + + fn layer_cache(&self) -> &'a ItemCache> { + self.layer_cache + } + + fn create_layer_target( + &mut self, + _item_rc: &ItemRc, + physical_size: euclid::Size2D, + ) -> Option { + let image_info = skia_safe::ImageInfo::new( + to_skia_size(&physical_size).to_ceil(), + skia_safe::ColorType::RGBA8888, + skia_safe::AlphaType::Premul, + None, + ); + self.canvas.new_surface(&image_info, None) + } + + fn render_into_layer( + &mut self, + mut surface: Self::LayerTarget, + item_rc: &ItemRc, + bounding_rect: LogicalRect, + ) -> Self::Image { + let canvas = surface.canvas(); + canvas.clear(skia_safe::Color::TRANSPARENT); + + let mut sub_renderer = SkiaItemRenderer::new( + canvas, + self.window, + self.surface, + self.image_cache, + self.layer_cache, + self.path_cache, + self.text_layout_cache, + self.box_shadow_cache, + ); + sub_renderer.translate(-bounding_rect.origin.to_vector()); + + i_slint_core::item_rendering::render_item_children( + &mut sub_renderer, + item_rc.item_tree(), + item_rc.index() as isize, + &WindowInner::from_pub(self.window).window_adapter(), + ); + + surface.image_snapshot() + } +} + +impl GlyphRenderer for SkiaItemRenderer<'_> { + type PlatformBrush = skia_safe::Paint; + + fn platform_text_fill_brush( + &mut self, + brush: i_slint_core::Brush, + size: LogicalSize, + ) -> Option { + self.brush_to_paint( + brush, + size.width_length() * self.scale_factor, + size.height_length() * self.scale_factor, + ) + } + + fn platform_brush_for_color( + &mut self, + color: &i_slint_core::Color, + ) -> Option { + if color.alpha() == 0 { + None + } else { + let mut paint = self.default_paint().unwrap_or_default(); + paint.set_shader(skia_safe::shaders::color(to_skia_color(color))); + Some(paint) + } + } + + fn platform_text_stroke_brush( + &mut self, + brush: i_slint_core::Brush, + physical_stroke_width: f32, + size: LogicalSize, + ) -> Option { + match self.brush_to_paint( + brush.clone(), + size.width_length() * self.scale_factor, + size.height_length() * self.scale_factor, + ) { + Some(mut stroke_paint) => { + stroke_paint.set_style(skia_safe::PaintStyle::Stroke); + stroke_paint.set_stroke_width(physical_stroke_width); + // Set stroke cap/join/miter to match FemtoVG + stroke_paint.set_stroke_cap(skia_safe::PaintCap::Butt); + stroke_paint.set_stroke_join(skia_safe::PaintJoin::Miter); + stroke_paint.set_stroke_miter(10.0); + Some(stroke_paint) + } + None => None, + } + } + + fn draw_glyph_run( + &mut self, + font: &sharedparley::parley::FontData, + font_size: PhysicalLength, + _normalized_coords: &[i16], + synthesis: &fontique::Synthesis, + brush: Self::PlatformBrush, + y_offset: sharedparley::PhysicalLength, + glyphs_it: &mut dyn Iterator, + ) { + let Some(type_face) = crate::font_cache::FONT_CACHE + .with_borrow_mut(|font_cache| font_cache.font_with_variations(font, synthesis)) + else { + return; + }; + let mut font = skia_safe::Font::from_typeface(type_face, font_size.get()); + font.set_subpixel(true); + + let (glyph_ids, glyph_positions): (Vec<_>, Vec<_>) = glyphs_it + .into_iter() + .map(|g| (g.id as skia_safe::GlyphId, skia_safe::Point::new(g.x, g.y + y_offset.get()))) + .unzip(); + + self.canvas.draw_glyphs_at( + &glyph_ids, + skia_safe::canvas::GlyphPositions::Points(&glyph_positions), + skia_safe::Point::default(), + &font, + &brush, + ); + } + + fn fill_rectangle( + &mut self, + physical_rect: sharedparley::PhysicalRect, + paint: Self::PlatformBrush, + ) { + self.canvas.draw_rect( + skia_safe::Rect::from_xywh( + physical_rect.min_x(), + physical_rect.min_y(), + physical_rect.width(), + physical_rect.height(), + ), + &paint, + ); + } +} + +pub fn from_skia_rect(rect: &skia_safe::Rect) -> PhysicalRect { + let top_left = euclid::Point2D::new(rect.left, rect.top); + let bottom_right = euclid::Point2D::new(rect.right, rect.bottom); + euclid::Box2D::new(top_left, bottom_right).to_rect() +} + +pub fn to_skia_rect(rect: &PhysicalRect) -> skia_safe::Rect { + skia_safe::Rect::from_xywh(rect.origin.x, rect.origin.y, rect.size.width, rect.size.height) +} + +pub fn to_skia_rrect(rect: &PhysicalRect, radius: &PhysicalBorderRadius) -> skia_safe::RRect { + if let Some(radius) = radius.as_uniform() { + skia_safe::RRect::new_rect_xy(to_skia_rect(rect), radius, radius) + } else { + skia_safe::RRect::new_rect_radii( + to_skia_rect(rect), + &[ + skia_safe::Point::new(radius.top_left, radius.top_left), + skia_safe::Point::new(radius.top_right, radius.top_right), + skia_safe::Point::new(radius.bottom_right, radius.bottom_right), + skia_safe::Point::new(radius.bottom_left, radius.bottom_left), + ], + ) + } +} + +impl ItemRendererFeatures for SkiaItemRenderer<'_> { + const SUPPORTS_TRANSFORMATIONS: bool = true; +} + +pub fn to_skia_point(point: PhysicalPoint) -> skia_safe::Point { + skia_safe::Point::new(point.x, point.y) +} + +pub fn to_skia_size(size: &PhysicalSize) -> skia_safe::Size { + skia_safe::Size::new(size.width, size.height) +} + +pub fn to_skia_color(col: &Color) -> skia_safe::Color { + skia_safe::Color::from_argb(col.alpha(), col.red(), col.green(), col.blue()) +} + +fn adjust_rect_and_border_for_inner_drawing( + rect: &mut PhysicalRect, + border_width: &mut PhysicalLength, +) { + // If the border width exceeds the width, just fill the rectangle. + *border_width = border_width.min(rect.width_length() / 2.); + // adjust the size so that the border is drawn within the geometry + + rect.origin += PhysicalSize::from_lengths(*border_width / 2., *border_width / 2.); + rect.size -= PhysicalSize::from_lengths(*border_width, *border_width); +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/lib.rs b/third_party/i-slint-renderer-skia-1.17.1/lib.rs new file mode 100644 index 0000000..3887109 --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/lib.rs @@ -0,0 +1,1154 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +// cSpell: ignore upcasting +#![doc = include_str!("README.md")] +#![doc(html_logo_url = "https://slint.dev/logo/slint-logo-square-light.svg")] +#![cfg_attr(slint_nightly_test, feature(non_exhaustive_omitted_patterns_lint))] +#![cfg_attr(slint_nightly_test, warn(non_exhaustive_omitted_patterns))] + +#[cfg(any(target_vendor = "apple", skia_backend_vulkan))] +use std::cell::OnceCell; +use std::cell::{Cell, RefCell}; +use std::pin::Pin; +use std::rc::{Rc, Weak}; +use std::sync::Arc; + +use i_slint_core::Brush; +use i_slint_core::api::{ + GraphicsAPI, PhysicalSize as PhysicalWindowSize, RenderingNotifier, RenderingState, + SetRenderingNotifierError, Window, +}; +use i_slint_core::graphics::RequestedGraphicsAPI; +use i_slint_core::graphics::euclid::{self, Vector2D}; +use i_slint_core::graphics::rendering_metrics_collector::RenderingMetricsCollector; +use i_slint_core::graphics::{BorderRadius, SharedPixelBuffer}; +use i_slint_core::item_rendering::{ItemCache, ItemRenderer}; +use i_slint_core::item_tree::ItemTreeWeak; +use i_slint_core::lengths::{ + LogicalLength, LogicalPoint, LogicalRect, LogicalSize, PhysicalPx, ScaleFactor, +}; +use i_slint_core::partial_renderer::{DirtyRegion, PartialRenderingState}; +use i_slint_core::platform::PlatformError; +use i_slint_core::renderer::DrawOutcome; +use i_slint_core::textlayout::sharedparley; +use i_slint_core::window::{WindowAdapter, WindowInner}; + +type PhysicalLength = euclid::Length; +type PhysicalRect = euclid::Rect; +type PhysicalSize = euclid::Size2D; +type PhysicalPoint = euclid::Point2D; +type PhysicalBorderRadius = BorderRadius; + +mod cached_image; +mod font_cache; +mod itemrenderer; + +#[cfg(skia_backend_software)] +pub mod software_surface; + +#[cfg(target_vendor = "apple")] +pub mod metal_surface; + +#[cfg(target_family = "windows")] +pub mod d3d_surface; + +#[cfg(skia_backend_vulkan)] +pub mod vulkan_surface; + +#[cfg(any(not(target_vendor = "apple"), target_os = "macos"))] +pub mod opengl_surface; + +#[cfg(feature = "wgpu-28")] +pub mod wgpu_28_surface; +#[cfg(feature = "wgpu-29")] +pub mod wgpu_29_surface; +#[cfg(feature = "wgpu-29")] +mod wgpu_renderer; +#[cfg(feature = "wgpu-29")] +pub use wgpu_renderer::SkiaWGPURenderer; + +use i_slint_core::items::{ItemRc, TextWrap}; +use itemrenderer::to_skia_rect; +pub use skia_safe; + +cfg_if::cfg_if! { + if #[cfg(skia_backend_vulkan)] { + type DefaultSurface = vulkan_surface::VulkanSurface; + } else if #[cfg(skia_backend_opengl)] { + type DefaultSurface = opengl_surface::OpenGLSurface; + } else if #[cfg(skia_backend_metal)] { + type DefaultSurface = metal_surface::MetalSurface; + } else if #[cfg(skia_backend_softbuffer)] { + type DefaultSurface = software_surface::SoftwareSurface; + } +} + +#[cfg(skia_windowed)] +fn create_default_surface( + context: &SkiaSharedContext, + window_handle: Arc, + display_handle: Arc, + size: PhysicalWindowSize, + requested_graphics_api: Option, +) -> Result, PlatformError> { + match DefaultSurface::new( + context, + window_handle.clone(), + display_handle.clone(), + size, + requested_graphics_api, + ) { + Ok(gpu_surface) => Ok(Box::new(gpu_surface) as Box), + #[cfg(skia_backend_softbuffer)] + Err(err) => { + i_slint_core::debug_log!( + "Failed to initialize Skia GPU renderer: {} . Falling back to software rendering", + err + ); + software_surface::SoftwareSurface::new( + context, + window_handle, + display_handle, + size, + None, + ) + .map(|r| Box::new(r) as Box) + } + #[cfg(not(skia_backend_softbuffer))] + Err(err) => Err(err), + } +} + +enum DirtyRegionDebugMode { + NoDebug, + Visualize, + Log, +} + +impl Default for DirtyRegionDebugMode { + fn default() -> Self { + match std::env::var("SLINT_SKIA_PARTIAL_RENDERING").as_deref() { + Ok("visualize") => DirtyRegionDebugMode::Visualize, + Ok("log") => DirtyRegionDebugMode::Log, + _ => DirtyRegionDebugMode::NoDebug, + } + } +} + +fn create_partial_renderer_state( + maybe_surface: Option<&dyn Surface>, +) -> Option { + maybe_surface + .map_or_else( + || std::env::var("SLINT_SKIA_PARTIAL_RENDERING").as_deref().is_ok(), + |surface| surface.use_partial_rendering(), + ) + .then(PartialRenderingState::default) +} + +#[derive(Default)] +struct SkiaSharedContextInner { + #[cfg(target_vendor = "apple")] + metal_context: OnceCell, + #[cfg(skia_backend_vulkan)] + vulkan_context: OnceCell, +} + +/// This data structure contains data that's intended to be shared across several instances of SkiaRenderer. +/// For example, for Vulkan rendering, this shares the Vulkan instance. +/// +/// Create an instance once and pass clones of it to the difference constructor functions, to ensure most +/// efficient resource usage. +#[derive(Clone, Default)] +pub struct SkiaSharedContext(#[allow(dead_code)] Rc); + +/// Use the SkiaRenderer when implementing a custom Slint platform where you deliver events to +/// Slint and want the scene to be rendered using Skia as underlying graphics library. +pub struct SkiaRenderer { + maybe_window_adapter: RefCell>>, + rendering_notifier: RefCell>>, + image_cache: ItemCache>, + layer_cache: ItemCache>, + path_cache: ItemCache, skia_safe::Path)>>, + text_layout_cache: sharedparley::TextLayoutCache, + rendering_metrics_collector: RefCell>>, + rendering_first_time: Cell, + surface: RefCell>>, + surface_factory: fn( + &SkiaSharedContext, + window_handle: Arc, + display_handle: Arc, + size: PhysicalWindowSize, + requested_graphics_api: Option, + ) -> Result, PlatformError>, + pre_present_callback: RefCell>>, + partial_rendering_state: Option, + dirty_region_debug_mode: DirtyRegionDebugMode, + /// Tracking dirty regions indexed by buffer age - 1. More than 3 back buffers aren't supported, but also unlikely to happen. + dirty_region_history: RefCell<[DirtyRegion; 3]>, + shared_context: SkiaSharedContext, +} + +impl SkiaRenderer { + #[cfg(skia_windowed)] + pub fn default(context: &SkiaSharedContext) -> Self { + Self { + maybe_window_adapter: Default::default(), + rendering_notifier: Default::default(), + image_cache: Default::default(), + layer_cache: Default::default(), + path_cache: Default::default(), + text_layout_cache: Default::default(), + rendering_metrics_collector: Default::default(), + rendering_first_time: Default::default(), + surface: Default::default(), + surface_factory: create_default_surface, + pre_present_callback: Default::default(), + partial_rendering_state: create_partial_renderer_state(None), + dirty_region_debug_mode: Default::default(), + dirty_region_history: Default::default(), + shared_context: context.clone(), + } + } + + #[cfg(skia_backend_software)] + /// Creates a new SkiaRenderer that will always use Skia's software renderer. + pub fn default_software(context: &SkiaSharedContext) -> Self { + Self { + maybe_window_adapter: Default::default(), + rendering_notifier: Default::default(), + image_cache: Default::default(), + layer_cache: Default::default(), + path_cache: Default::default(), + text_layout_cache: Default::default(), + rendering_metrics_collector: Default::default(), + rendering_first_time: Default::default(), + surface: Default::default(), + surface_factory: |context, + window_handle, + display_handle, + size, + requested_graphics_api| { + software_surface::SoftwareSurface::new( + context, + window_handle, + display_handle, + size, + requested_graphics_api, + ) + .map(|r| Box::new(r) as Box) + }, + pre_present_callback: Default::default(), + partial_rendering_state: PartialRenderingState::default().into(), + dirty_region_debug_mode: Default::default(), + dirty_region_history: Default::default(), + shared_context: context.clone(), + } + } + + #[cfg(any(not(target_vendor = "apple"), target_os = "macos"))] + /// Creates a new SkiaRenderer that will always use Skia's OpenGL renderer. + pub fn default_opengl(context: &SkiaSharedContext) -> Self { + Self { + maybe_window_adapter: Default::default(), + rendering_notifier: Default::default(), + image_cache: Default::default(), + layer_cache: Default::default(), + path_cache: Default::default(), + text_layout_cache: Default::default(), + rendering_metrics_collector: Default::default(), + rendering_first_time: Default::default(), + surface: Default::default(), + surface_factory: |context, + window_handle, + display_handle, + size, + requested_graphics_api| { + opengl_surface::OpenGLSurface::new( + context, + window_handle, + display_handle, + size, + requested_graphics_api, + ) + .map(|r| Box::new(r) as Box) + }, + pre_present_callback: Default::default(), + partial_rendering_state: create_partial_renderer_state(None), + dirty_region_debug_mode: Default::default(), + dirty_region_history: Default::default(), + shared_context: context.clone(), + } + } + + #[cfg(target_vendor = "apple")] + /// Creates a new SkiaRenderer that will always use Skia's Metal renderer. + pub fn default_metal(context: &SkiaSharedContext) -> Self { + Self { + maybe_window_adapter: Default::default(), + rendering_notifier: Default::default(), + image_cache: Default::default(), + layer_cache: Default::default(), + path_cache: Default::default(), + text_layout_cache: Default::default(), + rendering_metrics_collector: Default::default(), + rendering_first_time: Default::default(), + surface: Default::default(), + surface_factory: |context, + window_handle, + display_handle, + size, + requested_graphics_api| { + metal_surface::MetalSurface::new( + context, + window_handle, + display_handle, + size, + requested_graphics_api, + ) + .map(|r| Box::new(r) as Box) + }, + pre_present_callback: Default::default(), + partial_rendering_state: create_partial_renderer_state(None), + dirty_region_debug_mode: Default::default(), + dirty_region_history: Default::default(), + shared_context: context.clone(), + } + } + + #[cfg(skia_backend_vulkan)] + /// Creates a new SkiaRenderer that will always use Skia's Vulkan renderer. + pub fn default_vulkan(context: &SkiaSharedContext) -> Self { + Self { + maybe_window_adapter: Default::default(), + rendering_notifier: Default::default(), + image_cache: Default::default(), + layer_cache: Default::default(), + path_cache: Default::default(), + text_layout_cache: Default::default(), + rendering_metrics_collector: Default::default(), + rendering_first_time: Default::default(), + surface: Default::default(), + surface_factory: |context, + window_handle, + display_handle, + size, + requested_graphics_api| { + vulkan_surface::VulkanSurface::new( + context, + window_handle, + display_handle, + size, + requested_graphics_api, + ) + .map(|r| Box::new(r) as Box) + }, + pre_present_callback: Default::default(), + partial_rendering_state: create_partial_renderer_state(None), + dirty_region_debug_mode: Default::default(), + dirty_region_history: Default::default(), + shared_context: context.clone(), + } + } + + #[cfg(target_family = "windows")] + /// Creates a new SkiaRenderer that will always use Skia's Direct3D renderer. + pub fn default_direct3d(context: &SkiaSharedContext) -> Self { + Self { + maybe_window_adapter: Default::default(), + rendering_notifier: Default::default(), + image_cache: Default::default(), + layer_cache: Default::default(), + path_cache: Default::default(), + text_layout_cache: Default::default(), + rendering_metrics_collector: Default::default(), + rendering_first_time: Default::default(), + surface: Default::default(), + surface_factory: |context, + window_handle, + display_handle, + size, + requested_graphics_api| { + d3d_surface::D3DSurface::new( + context, + window_handle, + display_handle, + size, + requested_graphics_api, + ) + .map(|r| Box::new(r) as Box) + }, + pre_present_callback: Default::default(), + partial_rendering_state: create_partial_renderer_state(None), + dirty_region_debug_mode: Default::default(), + dirty_region_history: Default::default(), + shared_context: context.clone(), + } + } + + #[cfg(feature = "unstable-wgpu-28")] + /// Creates a new SkiaRenderer that will always use Skia's WGPU 28.x renderer. + pub fn default_wgpu_28(context: &SkiaSharedContext) -> Self { + Self { + maybe_window_adapter: Default::default(), + rendering_notifier: Default::default(), + image_cache: Default::default(), + layer_cache: Default::default(), + path_cache: Default::default(), + text_layout_cache: Default::default(), + rendering_metrics_collector: Default::default(), + rendering_first_time: Default::default(), + surface: Default::default(), + surface_factory: |context, + window_handle, + display_handle, + size, + requested_graphics_api| { + wgpu_28_surface::WGPUSurface::new( + context, + window_handle, + display_handle, + size, + requested_graphics_api, + ) + .map(|r| Box::new(r) as Box) + }, + pre_present_callback: Default::default(), + partial_rendering_state: create_partial_renderer_state(None), + dirty_region_debug_mode: Default::default(), + dirty_region_history: Default::default(), + shared_context: context.clone(), + } + } + + #[cfg(feature = "unstable-wgpu-29")] + /// Creates a new SkiaRenderer that will always use Skia's WGPU 29.x renderer. + pub fn default_wgpu_29(context: &SkiaSharedContext) -> Self { + Self { + maybe_window_adapter: Default::default(), + rendering_notifier: Default::default(), + image_cache: Default::default(), + layer_cache: Default::default(), + path_cache: Default::default(), + text_layout_cache: Default::default(), + rendering_metrics_collector: Default::default(), + rendering_first_time: Default::default(), + surface: Default::default(), + surface_factory: |context, + window_handle, + display_handle, + size, + requested_graphics_api| { + wgpu_29_surface::WGPUSurface::new( + context, + window_handle, + display_handle, + size, + requested_graphics_api, + ) + .map(|r| Box::new(r) as Box) + }, + pre_present_callback: Default::default(), + partial_rendering_state: create_partial_renderer_state(None), + dirty_region_debug_mode: Default::default(), + dirty_region_history: Default::default(), + shared_context: context.clone(), + } + } + + /// Creates a new renderer is associated with the provided window adapter. + pub fn new( + context: &SkiaSharedContext, + window_handle: Arc, + display_handle: Arc, + size: PhysicalWindowSize, + ) -> Result { + Ok(Self::new_with_surface( + context, + create_default_surface(context, window_handle, display_handle, size, None)?, + )) + } + + /// Creates a new renderer with the given surface trait implementation. + pub fn new_with_surface( + context: &SkiaSharedContext, + surface: Box, + ) -> Self { + let partial_rendering_state = create_partial_renderer_state(Some(surface.as_ref())); + Self { + maybe_window_adapter: Default::default(), + rendering_notifier: Default::default(), + image_cache: Default::default(), + layer_cache: Default::default(), + path_cache: Default::default(), + text_layout_cache: Default::default(), + rendering_metrics_collector: Default::default(), + rendering_first_time: Cell::new(true), + surface: RefCell::new(Some(surface)), + surface_factory: |_, _, _, _, _| { + Err("Skia renderer constructed with surface does not support dynamic surface re-creation".into()) + }, + pre_present_callback: Default::default(), + partial_rendering_state, + dirty_region_debug_mode: Default::default(), + dirty_region_history: Default::default(), + shared_context: context.clone(), + } + } + + /// Reset the surface to a new surface. (destroy the previously set surface if any) + pub fn set_surface(&self, surface: Box) { + self.image_cache.clear_all(); + self.path_cache.clear_all(); + self.text_layout_cache.clear_all(); + self.rendering_first_time.set(true); + *self.surface.borrow_mut() = Some(surface); + } + + fn clear_surface(&self) { + let Some(surface) = self.surface.borrow_mut().take() else { + return; + }; + + // If we've rendered a frame before, then we need to invoke the RenderingTearDown notifier. + if !self.rendering_first_time.get() + && let Some(callback) = self.rendering_notifier.borrow_mut().as_mut() + { + surface + .with_active_surface(&mut || { + surface.with_graphics_api(&mut |api| { + callback.notify(RenderingState::RenderingTeardown, &api) + }) + }) + .ok(); + } + + drop(surface); + } + + /// Suspends the renderer by freeing all graphics related resources as well as the underlying + /// rendering surface. Call [`Self::set_window_handle()`] to re-associate the renderer with a new + /// window surface for subsequent rendering. + pub fn suspend(&self) -> Result<(), PlatformError> { + self.image_cache.clear_all(); + self.path_cache.clear_all(); + self.text_layout_cache.clear_all(); + // Destroy the old surface before allocating the new one, to work around + // the vivante drivers using zwp_linux_explicit_synchronization_v1 and + // trying to create a second synchronization object and that's not allowed. + self.clear_surface(); + Ok(()) + } + + /// Reset the surface to the window given the window handle + pub fn set_window_handle( + &self, + window_handle: Arc, + display_handle: Arc, + size: PhysicalWindowSize, + requested_graphics_api: Option, + ) -> Result<(), PlatformError> { + // just in case + self.suspend()?; + let surface = (self.surface_factory)( + &self.shared_context, + window_handle, + display_handle, + size, + requested_graphics_api, + )?; + self.set_surface(surface); + Ok(()) + } + + /// Render the scene in the previously associated window. + pub fn render(&self) -> Result { + let window_adapter = self.window_adapter()?; + let size = window_adapter.window().size(); + self.internal_render_with_post_callback(0., (0., 0.), size, None) + } + + fn invoke_rendering_notifier_setup( + &self, + surface: &dyn Surface, + ) -> Result<(), i_slint_core::platform::PlatformError> { + if self.rendering_first_time.take() { + *self.rendering_metrics_collector.borrow_mut() = + RenderingMetricsCollector::new(&format!( + "Skia renderer (skia backend {}; surface: {} bpp)", + surface.name(), + surface.bits_per_pixel()? + )); + + if let Some(callback) = self.rendering_notifier.borrow_mut().as_mut() { + surface.with_graphics_api(&mut |api| { + callback.notify(RenderingState::RenderingSetup, &api) + }) + } + } + Ok(()) + } + + fn internal_render_with_post_callback( + &self, + rotation_angle_degrees: f32, + translation: (f32, f32), + surface_size: PhysicalWindowSize, + post_render_cb: Option<&dyn Fn(&mut dyn ItemRenderer)>, + ) -> Result { + let surface = self.surface.borrow(); + let Some(surface) = surface.as_ref() else { return Ok(DrawOutcome::Success) }; + self.invoke_rendering_notifier_setup(surface.as_ref())?; + + let window_adapter = self.window_adapter()?; + let window = window_adapter.window(); + + surface.render( + window, + surface_size, + &|skia_canvas, gr_context, back_buffer_age| { + self.render_to_canvas( + skia_canvas, + rotation_angle_degrees, + translation, + gr_context, + back_buffer_age, + Some(surface.as_ref()), + window, + post_render_cb, + ) + }, + &self.pre_present_callback, + ) + } + + fn render_to_canvas( + &self, + skia_canvas: &skia_safe::Canvas, + rotation_angle_degrees: f32, + translation: (f32, f32), + gr_context: Option<&mut skia_safe::gpu::DirectContext>, + back_buffer_age: u8, + surface: Option<&dyn Surface>, + window: &i_slint_core::api::Window, + post_render_cb: Option<&dyn Fn(&mut dyn ItemRenderer)>, + ) -> Option { + skia_canvas.rotate(rotation_angle_degrees, None); + skia_canvas.translate(translation); + + let window_inner = WindowInner::from_pub(window); + + let dirty_region = window_inner + .draw_contents(|components, post_render| { + self.render_components_to_canvas( + skia_canvas, + gr_context, + back_buffer_age, + surface, + window, + post_render_cb, + components, + post_render, + ) + }) + .unwrap_or_default(); + + if let Some(callback) = self.rendering_notifier.borrow_mut().as_mut() + && let Some(surface) = surface + { + surface + .with_graphics_api(&mut |api| callback.notify(RenderingState::AfterRendering, &api)) + } + + dirty_region + } + + fn render_components_to_canvas( + &self, + skia_canvas: &skia_safe::Canvas, + mut gr_context: Option<&mut skia_safe::gpu::DirectContext>, + back_buffer_age: u8, + surface: Option<&dyn Surface>, + window: &i_slint_core::api::Window, + post_render_cb: Option<&dyn Fn(&mut dyn ItemRenderer)>, + components: &[(ItemTreeWeak, LogicalPoint)], + post_render: &dyn Fn(&mut dyn ItemRenderer), + ) -> Option { + let window_inner = WindowInner::from_pub(window); + let window_adapter = window_inner.window_adapter(); + + let mut box_shadow_cache = Default::default(); + + self.image_cache.clear_cache_if_scale_factor_changed(window); + self.path_cache.clear_cache_if_scale_factor_changed(window); + self.text_layout_cache.clear_cache_if_scale_factor_changed(window); + + let mut skia_item_renderer = itemrenderer::SkiaItemRenderer::new( + skia_canvas, + window, + surface, + &self.image_cache, + &self.layer_cache, + &self.path_cache, + &self.text_layout_cache, + &mut box_shadow_cache, + ); + + let scale_factor = ScaleFactor::new(window_inner.scale_factor()); + let logical_window_size = i_slint_core::lengths::logical_size_from_api( + window.size().to_logical(window_inner.scale_factor()), + ); + + let mut dirty_region = None; + + { + let mut item_renderer: &mut dyn ItemRenderer = &mut skia_item_renderer; + let mut partial_renderer; + let mut dirty_region_to_visualize = None; + + if let Some(partial_rendering_state) = self.partial_rendering_state() { + partial_renderer = + partial_rendering_state.create_partial_renderer(skia_item_renderer); + + let mut dirty_region_history = self.dirty_region_history.borrow_mut(); + + let buffer_dirty_region = if back_buffer_age > 0 + && back_buffer_age as usize - 1 < dirty_region_history.len() + { + // The dirty region is the union of all the previous dirty regions + Some( + dirty_region_history[0..back_buffer_age as usize - 1] + .iter() + .fold(DirtyRegion::default(), |acc, region| acc.union(region)), + ) + } else { + Some(LogicalRect::from_size(logical_window_size).into()) + }; + + let dirty_region_for_this_frame = partial_rendering_state.apply_dirty_region( + &mut partial_renderer, + components, + logical_window_size, + buffer_dirty_region, + ); + + let mut clip_path_builder = skia_safe::PathBuilder::new(); + + for dirty_rect in partial_renderer.dirty_region.iter() { + let physical_rect = (dirty_rect * scale_factor).to_rect().round_out(); + clip_path_builder.add_rect(to_skia_rect(&physical_rect), None, None); + } + + let clip_path = clip_path_builder.detach(); + + if matches!(self.dirty_region_debug_mode, DirtyRegionDebugMode::Log) { + let area_to_repaint: f32 = + partial_renderer.dirty_region.iter().map(|b| b.area()).sum(); + i_slint_core::debug_log!( + "repainting {:.2}%", + area_to_repaint * 100. / logical_window_size.area() + ); + } + + dirty_region = partial_renderer.dirty_region.clone().into(); + + dirty_region_history.rotate_right(1); + dirty_region_history[0] = dirty_region_for_this_frame; + + skia_canvas.clip_path(&clip_path, None, false); + + if matches!(self.dirty_region_debug_mode, DirtyRegionDebugMode::Visualize) { + dirty_region_to_visualize = Some(clip_path); + } + + item_renderer = &mut partial_renderer; + } + + if let Some(window_item_rc) = window_inner.window_item_rc() { + let window_item = + window_item_rc.downcast::().unwrap(); + if let Brush::SolidColor(clear_color) = window_item.as_pin_ref().background() { + skia_canvas.clear(itemrenderer::to_skia_color(&clear_color)); + } else { + // Draws the window background as gradient + item_renderer.draw_rectangle( + window_item.as_pin_ref(), + &window_item_rc, + i_slint_core::lengths::logical_size_from_api( + window.size().to_logical(window_inner.scale_factor()), + ), + &window_item.as_pin_ref().cached_rendering_data, + ); + } + } + + if let Some(callback) = self.rendering_notifier.borrow_mut().as_mut() { + // For the BeforeRendering rendering notifier callback it's important that this happens *after* clearing + // the back buffer, in order to allow the callback to provide its own rendering of the background. + // Skia's clear() will merely schedule a clear call, so flush right away to make it immediate. + if let Some(ctx) = gr_context.as_mut() { + ctx.flush(None); + } + + if let Some(surface) = surface { + surface.with_graphics_api(&mut |api| { + callback.notify(RenderingState::BeforeRendering, &api) + }) + } + } + + for (component, origin) in components { + if let Some(component) = ItemTreeWeak::upgrade(component) { + i_slint_core::item_rendering::render_component_items( + &component, + item_renderer, + *origin, + &window_adapter, + ); + } + } + + post_render(item_renderer); + + if let Some(path) = dirty_region_to_visualize { + let mut paint = skia_safe::Paint::new( + skia_safe::Color4f { a: 0.5, r: 1.0, g: 0., b: 0. }, + None, + ); + paint.set_style(skia_safe::PaintStyle::Stroke); + skia_canvas.draw_path(&path, &paint); + } + + if let Some(collector) = &self.rendering_metrics_collector.borrow_mut().as_ref() { + collector.measure_frame_rendered(item_renderer, Default::default()); + if collector.refresh_mode() + == i_slint_core::graphics::rendering_metrics_collector::RefreshMode::FullSpeed + && let Some(partial_rendering_state) = self.partial_rendering_state() + { + partial_rendering_state.force_screen_refresh(); + } + } + + if let Some(cb) = post_render_cb.as_ref() { + cb(item_renderer) + } + } + + if let Some(ctx) = gr_context.as_mut() { + ctx.flush(None); + } + + dirty_region + } + + fn window_adapter(&self) -> Result, PlatformError> { + self.maybe_window_adapter.borrow().as_ref().and_then(|w| w.upgrade()).ok_or_else(|| { + "Renderer must be associated with component before use".to_string().into() + }) + } + + /// Sets the specified callback, that's invoked before presenting the rendered buffer to the windowing system. + /// This can be useful to implement frame throttling, i.e. for requesting a frame callback from the wayland compositor. + pub fn set_pre_present_callback(&self, callback: Option>) { + *self.pre_present_callback.borrow_mut() = callback; + } + + fn partial_rendering_state(&self) -> Option<&PartialRenderingState> { + // We don't know where the application might render to, so disable partial rendering. + if self.rendering_notifier.borrow().is_some() { + None + } else { + self.partial_rendering_state.as_ref() + } + } +} + +impl i_slint_core::renderer::RendererSealed for SkiaRenderer { + fn text_size( + &self, + text_item: Pin<&dyn i_slint_core::item_rendering::RenderString>, + item_rc: &ItemRc, + max_width: Option, + text_wrap: TextWrap, + ) -> LogicalSize { + sharedparley::text_size( + self, + text_item, + item_rc, + max_width, + text_wrap, + Some(&self.text_layout_cache), + ) + .unwrap_or_default() + } + + fn char_size( + &self, + text_item: Pin<&dyn i_slint_core::item_rendering::HasFont>, + item_rc: &i_slint_core::item_tree::ItemRc, + ch: char, + ) -> LogicalSize { + self.slint_context() + .and_then(|ctx| { + let mut font_ctx = ctx.font_context().borrow_mut(); + sharedparley::char_size(&mut font_ctx, text_item, item_rc, ch) + }) + .unwrap_or_default() + } + + fn font_metrics( + &self, + font_request: i_slint_core::graphics::FontRequest, + ) -> i_slint_core::items::FontMetrics { + self.slint_context() + .map(|ctx| { + let mut font_ctx = ctx.font_context().borrow_mut(); + sharedparley::font_metrics(&mut font_ctx, font_request) + }) + .unwrap_or_default() + } + + fn text_input_byte_offset_for_position( + &self, + text_input: std::pin::Pin<&i_slint_core::items::TextInput>, + item_rc: &i_slint_core::item_tree::ItemRc, + pos: LogicalPoint, + ) -> usize { + sharedparley::text_input_byte_offset_for_position(self, text_input, item_rc, pos) + } + + fn text_input_cursor_rect_for_byte_offset( + &self, + text_input: std::pin::Pin<&i_slint_core::items::TextInput>, + item_rc: &i_slint_core::item_tree::ItemRc, + byte_offset: usize, + ) -> LogicalRect { + sharedparley::text_input_cursor_rect_for_byte_offset(self, text_input, item_rc, byte_offset) + } + + fn register_font_from_memory( + &self, + data: &'static [u8], + ) -> Result<(), Box> { + let ctx = self.slint_context().ok_or("slint platform not initialized")?; + ctx.font_context().borrow_mut().register_static_font(data); + Ok(()) + } + + fn register_font_from_path( + &self, + path: &std::path::Path, + ) -> Result<(), Box> { + let requested_path = path.canonicalize().unwrap_or_else(|_| path.into()); + let contents = std::fs::read(requested_path)?; + let ctx = self.slint_context().ok_or("slint platform not initialized")?; + ctx.font_context().borrow_mut().collection.register_fonts(contents.into(), None); + Ok(()) + } + + fn set_rendering_notifier( + &self, + callback: Box, + ) -> std::result::Result<(), SetRenderingNotifierError> { + let mut notifier = self.rendering_notifier.borrow_mut(); + if notifier.replace(callback).is_some() { + Err(SetRenderingNotifierError::AlreadySet) + } else { + Ok(()) + } + } + + fn free_graphics_resources( + &self, + component: i_slint_core::item_tree::ItemTreeRef, + items: &mut dyn Iterator>>, + ) -> Result<(), i_slint_core::platform::PlatformError> { + self.image_cache.component_destroyed(component); + self.path_cache.component_destroyed(component); + self.text_layout_cache.component_destroyed(component); + + if let Some(partial_rendering_state) = self.partial_rendering_state() { + partial_rendering_state.free_graphics_resources(items); + } + + Ok(()) + } + + fn set_window_adapter(&self, window_adapter: &Rc) { + *self.maybe_window_adapter.borrow_mut() = Some(Rc::downgrade(window_adapter)); + self.image_cache.clear_all(); + self.path_cache.clear_all(); + self.text_layout_cache.clear_all(); + + if let Some(partial_rendering_state) = self.partial_rendering_state() { + partial_rendering_state.clear_cache(); + } + } + + fn window_adapter(&self) -> Option> { + self.maybe_window_adapter + .borrow() + .as_ref() + .and_then(|window_adapter| window_adapter.upgrade()) + } + + fn resize(&self, size: i_slint_core::api::PhysicalSize) -> Result<(), PlatformError> { + if size.width == 0 || size.height == 0 { + return Ok(()); + } + + if let Some(surface) = self.surface.borrow().as_ref() { + surface.resize_event(size) + } else { + Ok(()) + } + } + + /// Returns an image buffer of what was rendered last by reading the previous front buffer (using glReadPixels). + fn take_snapshot( + &self, + ) -> Result, PlatformError> { + let window_adapter = self.window_adapter()?; + let window = window_adapter.window(); + let size = window_adapter.window().size(); + let (width, height) = (size.width, size.height); + let mut target_buffer = + SharedPixelBuffer::::new(width, height); + + let mut surface_borrow = skia_safe::surfaces::wrap_pixels( + &skia_safe::ImageInfo::new( + (width as i32, height as i32), + skia_safe::ColorType::RGBA8888, + skia_safe::AlphaType::Opaque, + None, + ), + target_buffer.make_mut_bytes(), + None, + None, + ) + .ok_or_else(|| "Error wrapping target buffer for rendering into with Skia".to_string())?; + + self.render_to_canvas(surface_borrow.canvas(), 0., (0.0, 0.0), None, 0, None, window, None); + + Ok(target_buffer) + } + + fn mark_dirty_region(&self, region: DirtyRegion) { + if let Some(partial_rendering_state) = self.partial_rendering_state() { + partial_rendering_state.mark_dirty_region(region); + } + } + + fn supports_transformations(&self) -> bool { + true + } +} + +impl Drop for SkiaRenderer { + fn drop(&mut self) { + self.clear_surface() + } +} + +/// This trait represents the interface between the Skia renderer and the underlying rendering surface, such as a window +/// with a metal layer, a wayland window with an OpenGL context, etc. +pub trait Surface { + /// Creates a new surface with the given window, display, and size. + fn new( + shared_context: &SkiaSharedContext, + window_handle: Arc, + display_handle: Arc, + size: PhysicalWindowSize, + requested_graphics_api: Option, + ) -> Result + where + Self: Sized; + /// Returns the name of the surface, for diagnostic purposes. + fn name(&self) -> &'static str; + + /// If supported, this invokes the specified callback with access to the platform graphics API. + fn with_graphics_api(&self, _callback: &mut dyn FnMut(GraphicsAPI<'_>)) {} + /// Invokes the callback with the surface active. This has only a meaning for OpenGL rendering, where + /// the implementation must make the GL context current. + fn with_active_surface( + &self, + callback: &mut dyn FnMut(), + ) -> Result<(), i_slint_core::platform::PlatformError> { + callback(); + Ok(()) + } + /// Prepares the surface for rendering and invokes the provided callback with access to a Skia canvas and + /// rendering context. Returning `DrawOutcome::Occluded` or `Timeout` lets the caller + /// re-arm its redraw flag without rendering. + fn render( + &self, + window: &Window, + size: PhysicalWindowSize, + render_callback: &dyn Fn( + &skia_safe::Canvas, + Option<&mut skia_safe::gpu::DirectContext>, + u8, + ) -> Option, + pre_present_callback: &RefCell>>, + ) -> Result; + /// Called when the surface should be resized. + fn resize_event( + &self, + size: PhysicalWindowSize, + ) -> Result<(), i_slint_core::platform::PlatformError>; + fn bits_per_pixel(&self) -> Result; + + fn use_partial_rendering(&self) -> bool { + false + } + + fn import_opengl_texture( + &self, + _canvas: &skia_safe::Canvas, + _texture: &i_slint_core::graphics::BorrowedOpenGLTexture, + ) -> Option { + None + } + + #[cfg(any(feature = "unstable-wgpu-28", feature = "unstable-wgpu-29"))] + fn import_wgpu_texture( + &self, + _canvas: &skia_safe::Canvas, + _texture: &i_slint_core::graphics::WGPUTexture, + ) -> Option { + None + } + + /// Implementations should return self to allow upcasting. + fn as_any(&self) -> &dyn core::any::Any { + &() + } +} + +pub trait SkiaRendererExt { + fn render_transformed_with_post_callback( + &self, + rotation_angle_degrees: f32, + translation: (f32, f32), + surface_size: PhysicalWindowSize, + post_render_cb: Option<&dyn Fn(&mut dyn ItemRenderer)>, + ) -> Result; +} + +impl SkiaRendererExt for SkiaRenderer { + fn render_transformed_with_post_callback( + &self, + rotation_angle_degrees: f32, + translation: (f32, f32), + surface_size: PhysicalWindowSize, + post_render_cb: Option<&dyn Fn(&mut dyn ItemRenderer)>, + ) -> Result { + self.internal_render_with_post_callback( + rotation_angle_degrees, + translation, + surface_size, + post_render_cb, + ) + } +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/metal_surface.rs b/third_party/i-slint-renderer-skia-1.17.1/metal_surface.rs new file mode 100644 index 0000000..87b4972 --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/metal_surface.rs @@ -0,0 +1,270 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +// cSpell: ignore autoreleasepool drawables Snorm +use i_slint_core::api::{PhysicalSize as PhysicalWindowSize, Window}; +use i_slint_core::graphics::RequestedGraphicsAPI; +use i_slint_core::partial_renderer::DirtyRegion; +use i_slint_core::renderer::DrawOutcome; +use objc2::rc::autoreleasepool; +use objc2::{rc::Retained, runtime::ProtocolObject}; +use objc2_core_foundation::CGSize; +use objc2_metal::{MTLCommandBuffer, MTLCommandQueue, MTLDevice, MTLPixelFormat, MTLTexture}; +use objc2_quartz_core::{CAMetalDrawable, CAMetalLayer}; + +use skia_safe::gpu::mtl; + +use std::cell::RefCell; +use std::sync::Arc; + +use crate::SkiaSharedContext; + +pub struct SharedMetalContext { + device: Retained>, + command_queue: Retained>, +} + +impl super::SkiaSharedContextInner { + fn shared_metal_context( + &self, + ) -> Result<&SharedMetalContext, i_slint_core::platform::PlatformError> { + if let Some(ctx) = self.metal_context.get() { + return Ok(ctx); + } + self.metal_context.set(SharedMetalContext::new()?).ok(); + Ok(self.metal_context.get().unwrap()) + } +} + +impl SharedMetalContext { + fn new() -> Result { + let device = objc2_metal::MTLCreateSystemDefaultDevice().ok_or_else(|| { + "Skia Renderer: Unable to obtain metal system default device".to_string() + })?; + let command_queue = device + .newCommandQueue() + .ok_or_else(|| "Skia Renderer: Unable to create command queue".to_string())?; + Ok(Self { device, command_queue }) + } +} + +/// This surface renders into the given window using Metal. The provided display argument +/// is ignored, as it has no meaning on macOS. +pub struct MetalSurface { + command_queue: Retained>, + layer: raw_window_metal::Layer, + gr_context: RefCell, + // Map from drawable texture to age. Per https://developer.apple.com/documentation/quartzcore/cametallayer/maximumdrawablecount, CAMetalLayer + // can have either 2 or 3 drawables, but not more. That way, this vector is bound in growth. + drawable_ages: RefCell>, +} + +impl super::Surface for MetalSurface { + fn new( + shared_context: &SkiaSharedContext, + window_handle: Arc, + _display_handle: Arc, + size: PhysicalWindowSize, + requested_graphics_api: Option, + ) -> Result { + if requested_graphics_api.is_some_and(|api| !matches!(api, RequestedGraphicsAPI::Metal)) { + return Err("Requested non-Metal rendering with Metal renderer".into()); + } + + let layer = match window_handle + .window_handle() + .map_err(|e| format!("Error obtaining window handle for skia metal renderer: {e}"))? + .as_raw() + { + raw_window_handle::RawWindowHandle::AppKit(handle) => unsafe { + raw_window_metal::Layer::from_ns_view(handle.ns_view) + }, + raw_window_handle::RawWindowHandle::UiKit(handle) => unsafe { + raw_window_metal::Layer::from_ui_view(handle.ui_view) + }, + _ => return Err("Skia Renderer: Metal surface is only supported with AppKit".into()), + }; + + // SAFETY: The pointer is a valid `CAMetalLayer`. + let ca_layer: &CAMetalLayer = unsafe { layer.as_ptr().cast().as_ref() }; + + let shared_context = shared_context.0.shared_metal_context()?; + + let device = &shared_context.device; + + ca_layer.setDevice(Some(device)); + ca_layer.setPixelFormat(MTLPixelFormat::BGRA8Unorm); + ca_layer.setOpaque(false); + ca_layer.setPresentsWithTransaction(false); + + ca_layer.setDrawableSize(CGSize::new(size.width as f64, size.height as f64)); + + let flipped = ca_layer.contentsAreFlipped(); + let gravity = if !flipped { + unsafe { objc2_quartz_core::kCAGravityTopLeft } + } else { + unsafe { objc2_quartz_core::kCAGravityBottomLeft } + }; + ca_layer.setContentsGravity(gravity); + + let command_queue = shared_context.command_queue.clone(); + + let backend = unsafe { + mtl::BackendContext::new( + Retained::as_ptr(device) as mtl::Handle, + Retained::as_ptr(&command_queue) as mtl::Handle, + ) + }; + + let gr_context = + skia_safe::gpu::direct_contexts::make_metal(&backend, None).unwrap().into(); + + Ok(Self { command_queue, layer, gr_context, drawable_ages: Default::default() }) + } + + fn name(&self) -> &'static str { + "metal" + } + + fn resize_event( + &self, + size: PhysicalWindowSize, + ) -> Result<(), i_slint_core::platform::PlatformError> { + // SAFETY: The pointer is a valid `CAMetalLayer`. + let ca_layer: &CAMetalLayer = unsafe { self.layer.as_ptr().cast().as_ref() }; + ca_layer.setDrawableSize(CGSize::new(size.width as f64, size.height as f64)); + self.drawable_ages.borrow_mut().clear(); + Ok(()) + } + + fn render( + &self, + _window: &Window, + _size: PhysicalWindowSize, + callback: &dyn Fn( + &skia_safe::Canvas, + Option<&mut skia_safe::gpu::DirectContext>, + u8, + ) -> Option, + pre_present_callback: &RefCell>>, + ) -> Result { + autoreleasepool(|_| { + // SAFETY: The pointer is a valid `CAMetalLayer`. + let ca_layer: &CAMetalLayer = unsafe { self.layer.as_ptr().cast().as_ref() }; + let drawable = match ca_layer.nextDrawable() { + Some(drawable) => drawable, + None => { + return Err( + "Skia Metal Renderer: Failed to retrieve next drawable for rendering" + .into(), + ); + } + }; + + let gr_context = &mut self.gr_context.borrow_mut(); + + let size = ca_layer.drawableSize(); + + let mut surface = unsafe { + let texture = drawable.texture(); + let texture_info = mtl::TextureInfo::new(Retained::as_ptr(&texture) as mtl::Handle); + + let backend_render_target = skia_safe::gpu::backend_render_targets::make_mtl( + (size.width as i32, size.height as i32), + &texture_info, + ); + + skia_safe::gpu::surfaces::wrap_backend_render_target( + gr_context, + &backend_render_target, + skia_safe::gpu::SurfaceOrigin::TopLeft, + skia_safe::ColorType::BGRA8888, + None, + None, + ) + .unwrap() + }; + + let texture: Retained> = drawable.texture(); + let texture_id = texture.gpuResourceID(); + let age = { + let mut drawables = self.drawable_ages.borrow_mut(); + if let Some(existing_age) = + drawables.iter().find_map(|(id, age)| (*id == texture_id).then_some(*age)) + { + existing_age + } else { + drawables.push((texture_id, 0)); + 0 + } + }; + callback(surface.canvas(), Some(gr_context), age); + + drop(surface); + + gr_context.submit(None); + + if let Some(pre_present_callback) = pre_present_callback.borrow_mut().as_mut() { + pre_present_callback(); + } + + let command_buffer = self.command_queue.commandBuffer().ok_or_else(|| { + "Skia Renderer: Unable to obtain command queue's command buffer".to_string() + })?; + command_buffer.presentDrawable(ProtocolObject::from_ref(&*drawable)); + command_buffer.commit(); + + self.drawable_ages.borrow_mut().retain_mut(|(id, age)| { + if *id == texture_id { + *age = 1; + } else { + let Some(new_age) = age.checked_add(1) else { + // texture became too old, remove it. + return false; + }; + *age = new_age; + } + true + }); + + Ok(DrawOutcome::Success) + }) + } + + fn bits_per_pixel(&self) -> Result { + // SAFETY: The pointer is a valid `CAMetalLayer`. + let ca_layer: &CAMetalLayer = unsafe { self.layer.as_ptr().cast().as_ref() }; + + // From https://developer.apple.com/documentation/metal/mtlpixelformat: + // The storage size of each pixel format is determined by the sum of its components. + // For example, the storage size of BGRA8Unorm is 32 bits (four 8-bit components) and + // the storage size of BGR5A1Unorm is 16 bits (three 5-bit components and one 1-bit component). + Ok(match ca_layer.pixelFormat() { + MTLPixelFormat::B5G6R5Unorm + | MTLPixelFormat::A1BGR5Unorm + | MTLPixelFormat::ABGR4Unorm + | MTLPixelFormat::BGR5A1Unorm => 16, + MTLPixelFormat::RGBA8Unorm + | MTLPixelFormat::RGBA8Unorm_sRGB + | MTLPixelFormat::RGBA8Snorm + | MTLPixelFormat::RGBA8Uint + | MTLPixelFormat::RGBA8Sint + | MTLPixelFormat::BGRA8Unorm + | MTLPixelFormat::BGRA8Unorm_sRGB => 32, + MTLPixelFormat::RGB10A2Unorm + | MTLPixelFormat::RGB10A2Uint + | MTLPixelFormat::BGR10A2Unorm => 32, + MTLPixelFormat::RGBA16Unorm + | MTLPixelFormat::RGBA16Snorm + | MTLPixelFormat::RGBA16Uint + | MTLPixelFormat::RGBA16Sint => 64, + MTLPixelFormat::RGBA32Uint | MTLPixelFormat::RGBA32Sint => 128, + fmt => { + return Err(format!( + "Skia Metal Renderer: Unsupported layer pixel format found {fmt:?}" + ) + .into()); + } + }) + } +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/opengl_surface.rs b/third_party/i-slint-renderer-skia-1.17.1/opengl_surface.rs new file mode 100644 index 0000000..56367c6 --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/opengl_surface.rs @@ -0,0 +1,531 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +// cSpell: ignore fboid +use std::num::NonZeroU32; +use std::{cell::RefCell, sync::Arc}; + +use glutin::{ + config::GetGlConfig, + context::{ContextApi, ContextAttributesBuilder}, + display::GetGlDisplay, + prelude::*, + surface::{SurfaceAttributesBuilder, WindowSurface}, +}; +use i_slint_core::api::{GraphicsAPI, PhysicalSize as PhysicalWindowSize, Window}; +use i_slint_core::graphics::{BorrowedOpenGLTexture, RequestedGraphicsAPI, RequestedOpenGLVersion}; +use i_slint_core::partial_renderer::DirtyRegion; +use i_slint_core::platform::PlatformError; +use i_slint_core::renderer::DrawOutcome; + +use crate::SkiaSharedContext; + +/// Wraps a [`glutin::context::PossiblyCurrentContext`] and makes it not-current on drop, +/// so that no stale thread-local state remains after the context is destroyed. +struct GlutinContext(Option); + +impl GlutinContext { + fn new(context: glutin::context::PossiblyCurrentContext) -> Self { + Self(Some(context)) + } +} + +impl std::ops::Deref for GlutinContext { + type Target = glutin::context::PossiblyCurrentContext; + fn deref(&self) -> &Self::Target { + self.0.as_ref().unwrap() + } +} + +impl Drop for GlutinContext { + fn drop(&mut self) { + if let Some(context) = self.0.take() + && let Err(e) = context.make_not_current() + { + i_slint_core::debug_log!( + "Skia OpenGL Renderer: Failed to make context not current: {e}" + ); + } + } +} + +/// This surface type renders into the given window with OpenGL, using glutin and glow libraries. +pub struct OpenGLSurface { + fb_info: skia_safe::gpu::gl::FramebufferInfo, + surface: RefCell, + gr_context: RefCell, + glutin_context: GlutinContext, + glutin_surface: glutin::surface::Surface, +} + +impl super::Surface for OpenGLSurface { + fn new( + _shared_context: &SkiaSharedContext, + window_handle: Arc, + display_handle: Arc, + size: PhysicalWindowSize, + requested_graphics_api: Option, + ) -> Result { + Self::new_with_config( + window_handle, + display_handle, + size, + requested_graphics_api.as_ref().map(TryInto::try_into).transpose()?, + glutin::config::ConfigTemplateBuilder::new(), + None, + ) + } + + fn name(&self) -> &'static str { + "opengl" + } + + fn with_graphics_api(&self, callback: &mut dyn FnMut(GraphicsAPI<'_>)) { + let api = GraphicsAPI::NativeOpenGL { + get_proc_address: &|name| { + self.glutin_context.display().get_proc_address(name) as *const _ + }, + }; + callback(api) + } + + fn with_active_surface(&self, callback: &mut dyn FnMut()) -> Result<(), PlatformError> { + self.ensure_context_current()?; + callback(); + Ok(()) + } + + fn render( + &self, + _window: &Window, + size: PhysicalWindowSize, + callback: &dyn Fn( + &skia_safe::Canvas, + Option<&mut skia_safe::gpu::DirectContext>, + u8, + ) -> Option, + pre_present_callback: &RefCell>>, + ) -> Result { + self.ensure_context_current()?; + + let current_context = &self.glutin_context; + + let gr_context = &mut self.gr_context.borrow_mut(); + + let mut surface = self.surface.borrow_mut(); + + let width = size.width.try_into().ok(); + let height = size.height.try_into().ok(); + + if let Some((width, height)) = width.zip(height) + && (width != surface.width() || height != surface.height()) + { + *surface = Self::create_internal_surface( + self.fb_info, + current_context, + gr_context, + width, + height, + )?; + } + + let skia_canvas = surface.canvas(); + + skia_canvas.save(); + callback( + skia_canvas, + Some(gr_context), + u8::try_from(self.glutin_surface.buffer_age()).unwrap_or_default(), + ); + skia_canvas.restore(); + + if let Some(pre_present_callback) = pre_present_callback.borrow_mut().as_mut() { + pre_present_callback(); + } + + self.glutin_surface.swap_buffers(current_context).map(|_| DrawOutcome::Success).map_err( + |glutin_error| { + format!("Skia OpenGL Renderer: Error swapping buffers: {glutin_error}").into() + }, + ) + } + + fn resize_event(&self, size: PhysicalWindowSize) -> Result<(), PlatformError> { + self.ensure_context_current()?; + + if let Some((width, height)) = size.width.try_into().ok().zip(size.height.try_into().ok()) { + self.glutin_surface.resize(&self.glutin_context, width, height); + } + + Ok(()) + } + + fn bits_per_pixel(&self) -> Result { + let config = self.glutin_context.config(); + let rgb_bits = match config.color_buffer_type() { + Some(glutin::config::ColorBufferType::Rgb { r_size, g_size, b_size }) => { + r_size + g_size + b_size + } + other => { + return Err(format!( + "Skia OpenGL Renderer: unsupported color buffer {other:?} encountered" + ) + .into()); + } + }; + Ok(rgb_bits + config.alpha_size()) + } + + fn import_opengl_texture( + &self, + canvas: &skia_safe::Canvas, + BorrowedOpenGLTexture { texture_id, size, origin, .. }: &BorrowedOpenGLTexture, + ) -> Option { + unsafe { + let mut texture_info = skia_safe::gpu::gl::TextureInfo::from_target_and_id( + glow::TEXTURE_2D, + texture_id.get(), + ); + texture_info.format = glow::RGBA8; + let backend_texture = skia_safe::gpu::backend_textures::make_gl( + (size.width as _, size.height as _), + skia_safe::gpu::Mipmapped::No, + texture_info, + "Borrowed GL texture", + ); + skia_safe::image::Image::from_texture( + canvas.recording_context().as_mut().unwrap(), + &backend_texture, + match origin { + i_slint_core::graphics::BorrowedOpenGLTextureOrigin::TopLeft => { + skia_safe::gpu::SurfaceOrigin::TopLeft + } + i_slint_core::graphics::BorrowedOpenGLTextureOrigin::BottomLeft => { + skia_safe::gpu::SurfaceOrigin::BottomLeft + } + _ => unimplemented!( + "internal error: missing implementation for BorrowedOpenGLTextureOrigin" + ), + }, + skia_safe::ColorType::RGBA8888, + skia_safe::AlphaType::Unpremul, + None, + ) + } + } +} + +impl OpenGLSurface { + pub fn new_with_config( + window_handle: Arc, + display_handle: Arc, + size: PhysicalWindowSize, + requested_opengl_version: Option, + config_builder: glutin::config::ConfigTemplateBuilder, + config_filter: Option<&dyn Fn(&glutin::config::Config) -> bool>, + ) -> Result { + let width: std::num::NonZeroU32 = size.width.try_into().map_err(|_| { + format!("Attempting to create window surface with an invalid width: {}", size.width) + })?; + let height: std::num::NonZeroU32 = size.height.try_into().map_err(|_| { + format!("Attempting to create window surface with an invalid height: {}", size.height) + })?; + + let window_handle = window_handle + .window_handle() + .map_err(|e| format!("error obtaining window handle for skia opengl renderer: {e}"))?; + let display_handle = display_handle + .display_handle() + .map_err(|e| format!("error obtaining display handle for skia opengl renderer: {e}"))?; + + let (current_glutin_context, glutin_surface) = Self::init_glutin( + window_handle, + display_handle, + width, + height, + requested_opengl_version, + config_builder, + config_filter, + )?; + + glutin_surface.resize(¤t_glutin_context, width, height); + + let fb_info = { + use glow::HasContext; + + let gl = unsafe { + glow::Context::from_loader_function_cstr(|name| { + current_glutin_context.display().get_proc_address(name) as *const _ + }) + }; + let fboid = unsafe { gl.get_parameter_i32(glow::FRAMEBUFFER_BINDING) }; + + skia_safe::gpu::gl::FramebufferInfo { + fboid: fboid.try_into().map_err(|_| { + "Skia Renderer: Internal error, framebuffer binding returned signed id" + .to_string() + })?, + format: skia_safe::gpu::gl::Format::RGBA8.into(), + ..Default::default() + } + }; + + let gl_interface = skia_safe::gpu::gl::Interface::new_load_with_cstr(|name| { + current_glutin_context.display().get_proc_address(name) as *const _ + }) + .ok_or_else(|| { + "Skia Renderer: Internal Error: Could not create OpenGL Interface".to_string() + })?; + + let mut gr_context = + skia_safe::gpu::direct_contexts::make_gl(gl_interface, None).ok_or_else(|| { + "Skia Renderer: Internal Error: Could not create Skia Direct Context from GL interface".to_string() + })?; + + let width: i32 = size.width.try_into().map_err(|e| { + format!("Attempting to create window surface with width that doesn't fit into non-zero i32: {e}") + })?; + let height: i32 = size.height.try_into().map_err(|e| { + format!( + "Attempting to create window surface with height that doesn't fit into non-zero i32: {e}" + ) + })?; + + let surface = Self::create_internal_surface( + fb_info, + ¤t_glutin_context, + &mut gr_context, + width, + height, + )? + .into(); + + Ok(Self { + fb_info, + surface, + gr_context: RefCell::new(gr_context), + glutin_context: GlutinContext::new(current_glutin_context), + glutin_surface, + }) + } + + fn init_glutin( + _window_handle: raw_window_handle::WindowHandle<'_>, + _display_handle: raw_window_handle::DisplayHandle<'_>, + width: NonZeroU32, + height: NonZeroU32, + requested_opengl_version: Option, + config_template_builder: glutin::config::ConfigTemplateBuilder, + config_filter: Option<&dyn Fn(&glutin::config::Config) -> bool>, + ) -> Result< + ( + glutin::context::PossiblyCurrentContext, + glutin::surface::Surface, + ), + PlatformError, + > { + cfg_if::cfg_if! { + if #[cfg(target_os = "macos")] { + let display_api_preference = glutin::display::DisplayApiPreference::Cgl; + } else if #[cfg(not(target_family = "windows"))] { + let display_api_preference = glutin::display::DisplayApiPreference::Egl; + } else { + let display_api_preference = glutin::display::DisplayApiPreference::EglThenWgl(Some(_window_handle.as_raw())); + } + } + + let gl_display = unsafe { + glutin::display::Display::new(_display_handle.as_raw(), display_api_preference) + .map_err(|glutin_error| { + format!( + "Error creating glutin display for native display {:?}: {}", + _display_handle.as_raw(), + glutin_error + ) + })? + }; + + // On macOS, there's only one GL config and that's initialized based on the values in the config template + // builder. So if that one has transparency enabled, it'll show up in the config, and will be set on the + // context later. So we must enable it here, there's no way of enabling it later. + // On EGL/GLX/WGL there are system provided configs that may or may not support transparency. Here in case + // the system doesn't support transparency, we want to fall back to a config that doesn't - better than not + // rendering anything at all. So we don't want to limit the configurations we get to see early on. + // Commented out due to https://github.com/rust-windowing/glutin/issues/1640 + #[cfg(target_os = "macos")] + let config_template_builder = config_template_builder.with_transparency(true); + + // Upstream advises to use this only on Windows. + #[cfg(target_family = "windows")] + let config_template_builder = + config_template_builder.compatible_with_native_window(_window_handle.as_raw()); + + let config_template = config_template_builder.build(); + + let config = unsafe { + gl_display + .find_configs(config_template) + .map_err(|e| format!("Could not find valid OpenGL display configurations: {e}"))? + .filter(|config| config_filter.as_ref().is_none_or(|filter_fn| filter_fn(config))) + .reduce(|accum, config| { + let transparency_check = config.supports_transparency().unwrap_or(false) + & !accum.supports_transparency().unwrap_or(false); + + if transparency_check || config.num_samples() < accum.num_samples() { + config + } else { + accum + } + }) + .ok_or("Unable to find suitable GL config")? + }; + + let requested_opengl_version = + requested_opengl_version.unwrap_or(RequestedOpenGLVersion::OpenGLES(Some((3, 0)))); + let preferred_context_attributes = match requested_opengl_version { + RequestedOpenGLVersion::OpenGL(version) => { + let version = + version.map(|(major, minor)| glutin::context::Version { major, minor }); + ContextAttributesBuilder::new() + .with_context_api(ContextApi::OpenGl(version)) + .build(Some(_window_handle.as_raw())) + } + RequestedOpenGLVersion::OpenGLES(version) => { + let version = + version.map(|(major, minor)| glutin::context::Version { major, minor }); + + ContextAttributesBuilder::new() + .with_context_api(ContextApi::Gles(version)) + .build(Some(_window_handle.as_raw())) + } + }; + + let gles2_fallback_context_attributes = ContextAttributesBuilder::new() + .with_context_api(ContextApi::Gles(Some(glutin::context::Version { + major: 2, + minor: 0, + }))) + .build(Some(_window_handle.as_raw())); + + let fallback_context_attributes = + ContextAttributesBuilder::new().build(Some(_window_handle.as_raw())); + + let not_current_gl_context = unsafe { + gl_display + .create_context(&config, &preferred_context_attributes) + .or_else(|_| gl_display.create_context(&config, &gles2_fallback_context_attributes)) + .or_else(|_| gl_display.create_context(&config, &fallback_context_attributes)) + .map_err(|e| format!("Error creating OpenGL context: {e}")) + }?; + + let attrs = SurfaceAttributesBuilder::::new().build( + _window_handle.as_raw(), + width, + height, + ); + + let surface = unsafe { + config + .display() + .create_window_surface(&config, &attrs) + .map_err(|e| format!("Error creating OpenGL window surface: {e}"))? + }; + + let context = not_current_gl_context.make_current(&surface) + .map_err(|glutin_error: glutin::error::Error| -> PlatformError { + format!("FemtoVG Renderer: Failed to make newly created OpenGL context current: {glutin_error}") + .into() + })?; + + // Align the GL layer to the top-left, so that resizing only invalidates the bottom/right + // part of the window. + #[cfg(target_os = "macos")] + if let raw_window_handle::RawWindowHandle::AppKit(raw_window_handle::AppKitWindowHandle { + ns_view, + .. + }) = _window_handle.as_raw() + { + let ns_view: &objc2_app_kit::NSView = unsafe { ns_view.cast().as_ref() }; + ns_view.setLayerContentsPlacement(objc2_app_kit::NSViewLayerContentsPlacement::TopLeft); + } + + // Sanity check, as all this might succeed on Windows without working GL drivers, but this will fail: + if context + .display() + .get_proc_address(&std::ffi::CString::new("glCreateShader").unwrap()) + .is_null() + { + return Err( + "Failed to initialize OpenGL driver: Could not locate glCreateShader symbol" + .to_string() + .into(), + ); + } + + // Try to default to vsync and ignore if the driver doesn't support it. + surface + .set_swap_interval( + &context, + glutin::surface::SwapInterval::Wait(NonZeroU32::new(1).unwrap()), + ) + .ok(); + + Ok((context, surface)) + } + + fn create_internal_surface( + fb_info: skia_safe::gpu::gl::FramebufferInfo, + gl_context: &glutin::context::PossiblyCurrentContext, + gr_context: &mut skia_safe::gpu::DirectContext, + width: i32, + height: i32, + ) -> Result { + let config = gl_context.config(); + + let backend_render_target = skia_safe::gpu::backend_render_targets::make_gl( + (width, height), + Some(config.num_samples() as _), + config.stencil_size() as _, + fb_info, + ); + match skia_safe::gpu::surfaces::wrap_backend_render_target( + gr_context, + &backend_render_target, + skia_safe::gpu::SurfaceOrigin::BottomLeft, + skia_safe::ColorType::RGBA8888, + None, + None, + ) { + Some(surface) => Ok(surface), + None => { + Err("Skia OpenGL Renderer: Failed to allocate internal backend rendering target" + .into()) + } + } + } + + fn ensure_context_current(&self) -> Result<(), PlatformError> { + if !self.glutin_context.is_current() { + self.glutin_context.make_current(&self.glutin_surface).map_err( + |glutin_error| -> PlatformError { + format!("Skia Renderer: Error making context current: {glutin_error}").into() + }, + )?; + } + Ok(()) + } +} + +impl Drop for OpenGLSurface { + fn drop(&mut self) { + // Make sure that the context is current before Skia calls glDelete*** + // In the event that this fails for some reason (lost GL context), convey that to Skia so that it doesn't try to call + // glDelete*** + if self.ensure_context_current().is_err() { + i_slint_core::debug_log!( + "Skia OpenGL Renderer warning: Failed to make context current for destruction - considering context abandoned." + ); + self.gr_context.borrow_mut().abandon(); + } + } +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/software_surface.rs b/third_party/i-slint-renderer-skia-1.17.1/software_surface.rs new file mode 100644 index 0000000..e72e51f --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/software_surface.rs @@ -0,0 +1,256 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +use i_slint_core::api::{PhysicalSize as PhysicalWindowSize, Window}; +use i_slint_core::graphics::RequestedGraphicsAPI; +use i_slint_core::partial_renderer::DirtyRegion; +use i_slint_core::renderer::DrawOutcome; + +use std::cell::RefCell; +use std::num::NonZeroU32; +use std::rc::Rc; +use std::sync::Arc; + +use crate::SkiaSharedContext; + +pub trait RenderBuffer { + fn with_buffer( + &self, + window: &Window, + size: PhysicalWindowSize, + render_callback: &mut dyn FnMut( + NonZeroU32, + NonZeroU32, + skia_safe::ColorType, + u8, + &mut [u8], + ) -> Result< + Option, + i_slint_core::platform::PlatformError, + >, + ) -> Result<(), i_slint_core::platform::PlatformError>; +} + +#[cfg(feature = "softbuffer")] +struct SoftbufferRenderBuffer { + _context: softbuffer::Context>, + surface: RefCell< + softbuffer::Surface< + Arc, + Arc, + >, + >, +} + +#[cfg(feature = "softbuffer")] +impl RenderBuffer for SoftbufferRenderBuffer { + fn with_buffer( + &self, + window: &Window, + size: PhysicalWindowSize, + render_callback: &mut dyn FnMut( + NonZeroU32, + NonZeroU32, + skia_safe::ColorType, + u8, + &mut [u8], + ) -> Result< + Option, + i_slint_core::platform::PlatformError, + >, + ) -> Result<(), i_slint_core::platform::PlatformError> { + let Some((width, height)) = size.width.try_into().ok().zip(size.height.try_into().ok()) + else { + // Nothing to render + return Ok(()); + }; + + let mut surface = self.surface.borrow_mut(); + + surface + .resize(width, height) + .map_err(|e| format!("Error resizing softbuffer surface: {e}"))?; + + let mut target_buffer = surface + .buffer_mut() + .map_err(|e| format!("Error retrieving softbuffer rendering buffer: {e}"))?; + + let dirty_region = render_callback( + width, + height, + skia_safe::ColorType::BGRA8888, + target_buffer.age(), + bytemuck::cast_slice_mut(target_buffer.as_mut()), + )?; + + if let Some(dirty_region) = dirty_region { + let scale_factor = i_slint_core::lengths::ScaleFactor::new(window.scale_factor()); + + let damage_rects = dirty_region + .iter() + .map(|logical| { + let physical_rect = (logical.to_rect() * scale_factor).round_out(); + softbuffer::Rect { + x: physical_rect.min_x().ceil() as _, + y: physical_rect.min_y().ceil() as _, + width: ((physical_rect.width() as i32).max(1) as u32).try_into().unwrap(), + height: ((physical_rect.height() as i32).max(1) as u32).try_into().unwrap(), + } + }) + .collect::>(); + target_buffer.present_with_damage(&damage_rects) + } else { + target_buffer.present() + } + .map_err(|e| format!("Error presenting softbuffer buffer after skia rendering: {e}"))?; + + Ok(()) + } +} + +/// This surface renders into the given window using Skia's software rasterize. +pub struct SoftwareSurface { + render_buffer: Box, +} + +impl super::Surface for SoftwareSurface { + #[cfg(feature = "softbuffer")] + fn new( + _shared_context: &SkiaSharedContext, + window_handle: Arc, + display_handle: Arc, + _size: PhysicalWindowSize, + _requested_graphics_api: Option, + ) -> Result { + let _context = softbuffer::Context::new(display_handle) + .map_err(|e| format!("Error creating softbuffer context: {e}"))?; + + let surface = + softbuffer::Surface::new(&_context, window_handle).map_err(|softbuffer_error| { + format!("Error creating softbuffer surface: {softbuffer_error}") + })?; + + let surface_access = + Box::new(SoftbufferRenderBuffer { _context, surface: RefCell::new(surface) }); + + Ok(Self { render_buffer: surface_access }) + } + + #[cfg(not(feature = "softbuffer"))] + fn new( + _shared_context: &SkiaSharedContext, + _window_handle: Arc, + _display_handle: Arc, + _size: PhysicalWindowSize, + _requested_graphics_api: Option, + ) -> Result { + struct DummyBuffer; + impl RenderBuffer for DummyBuffer { + fn with_buffer( + &self, + _window: &Window, + _size: PhysicalWindowSize, + _render_callback: &mut dyn FnMut( + std::num::NonZeroU32, + std::num::NonZeroU32, + skia_safe::ColorType, + u8, + &mut [u8], + ) -> Result< + Option, + i_slint_core::platform::PlatformError, + >, + ) -> Result<(), i_slint_core::platform::PlatformError> { + Err("Slint's Skia renderer compiled without the 'softbuffer' feature cannot render into a window".into()) + } + } + Ok(DummyBuffer.into()) + } + + fn name(&self) -> &'static str { + "software" + } + + fn resize_event( + &self, + _size: PhysicalWindowSize, + ) -> Result<(), i_slint_core::platform::PlatformError> { + Ok(()) + } + + fn render( + &self, + window: &Window, + size: PhysicalWindowSize, + callback: &dyn Fn( + &skia_safe::Canvas, + Option<&mut skia_safe::gpu::DirectContext>, + u8, + ) -> Option, + pre_present_callback: &RefCell>>, + ) -> Result { + self.render_buffer.with_buffer( + window, + size, + &mut |width, height, pixel_format, age, pixels| { + let mut surface_borrow = skia_safe::surfaces::wrap_pixels( + &skia_safe::ImageInfo::new( + (width.get() as i32, height.get() as i32), + pixel_format, + skia_safe::AlphaType::Opaque, + None, + ), + pixels, + None, + None, + ) + .ok_or_else(|| { + "Error wrapping target buffer for rendering into with Skia".to_string() + })?; + + let dirty_region = callback(surface_borrow.canvas(), None, age); + + if let Some(pre_present_callback) = pre_present_callback.borrow_mut().as_mut() { + pre_present_callback(); + } + + Ok(dirty_region) + }, + )?; + Ok(DrawOutcome::Success) + } + + fn bits_per_pixel(&self) -> Result { + Ok(24) + } + + fn use_partial_rendering(&self) -> bool { + true + } +} + +impl From for SoftwareSurface { + fn from(render_buffer: T) -> Self { + Self { render_buffer: Box::new(render_buffer) } + } +} + +impl RenderBuffer for Rc { + fn with_buffer( + &self, + window: &Window, + size: PhysicalWindowSize, + render_callback: &mut dyn FnMut( + NonZeroU32, + NonZeroU32, + skia_safe::ColorType, + u8, + &mut [u8], + ) -> Result< + Option, + i_slint_core::platform::PlatformError, + >, + ) -> Result<(), i_slint_core::platform::PlatformError> { + self.as_ref().with_buffer(window, size, render_callback) + } +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/vulkan_surface.rs b/third_party/i-slint-renderer-skia-1.17.1/vulkan_surface.rs new file mode 100644 index 0000000..97023bc --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/vulkan_surface.rs @@ -0,0 +1,525 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +// cSpell: ignore madsmtm +use std::cell::{Cell, RefCell}; +use std::sync::Arc; + +use i_slint_core::api::{PhysicalSize as PhysicalWindowSize, Window}; +use i_slint_core::graphics::RequestedGraphicsAPI; +use i_slint_core::partial_renderer::DirtyRegion; +use i_slint_core::renderer::DrawOutcome; + +use vulkano::device::physical::{PhysicalDevice, PhysicalDeviceType}; +use vulkano::device::{ + Device, DeviceCreateInfo, DeviceExtensions, Queue, QueueCreateInfo, QueueFlags, +}; +use vulkano::image::view::ImageView; +use vulkano::image::{Image, ImageUsage}; +use vulkano::instance::{Instance, InstanceCreateFlags, InstanceCreateInfo, InstanceExtensions}; +use vulkano::swapchain::{Surface, Swapchain, SwapchainCreateInfo, SwapchainPresentInfo}; +use vulkano::sync::GpuFuture; +use vulkano::{Handle, Validated, VulkanError, VulkanLibrary, VulkanObject, sync}; + +use crate::SkiaSharedContext; + +pub struct SharedVulkanContext { + instance: Arc, + // TODO: share also physical/logical device and queue, but their selection process is surface compatibility dependent. +} + +impl super::SkiaSharedContextInner { + fn shared_vulkan_context( + &self, + ) -> Result<&SharedVulkanContext, i_slint_core::platform::PlatformError> { + if let Some(ctx) = self.vulkan_context.get() { + return Ok(ctx); + } + self.vulkan_context.set(SharedVulkanContext::new()?).ok(); + Ok(self.vulkan_context.get().unwrap()) + } +} + +impl SharedVulkanContext { + fn new() -> Result { + let library = VulkanLibrary::new() + .map_err(|load_err| format!("Error loading vulkan library: {load_err}"))?; + + let required_extensions = InstanceExtensions { + khr_surface: true, + mvk_macos_surface: true, + ext_metal_surface: true, + khr_wayland_surface: true, + khr_xlib_surface: true, + khr_xcb_surface: true, + khr_win32_surface: true, + khr_get_surface_capabilities2: true, + khr_get_physical_device_properties2: true, + ..InstanceExtensions::empty() + } + .intersection(library.supported_extensions()); + + let instance = Instance::new( + library.clone(), + InstanceCreateInfo { + flags: InstanceCreateFlags::ENUMERATE_PORTABILITY, + enabled_extensions: required_extensions, + ..Default::default() + }, + ) + .map_err(|instance_err| format!("Error creating Vulkan instance: {instance_err}"))?; + Ok(Self { instance }) + } +} + +/// This surface renders into the given window using Vulkan. +pub struct VulkanSurface { + gr_context: RefCell, + recreate_swapchain: Cell, + device: Arc, + previous_frame_end: RefCell>>, + queue: Arc, + swapchain: RefCell>, + swapchain_images: RefCell>>, + swapchain_image_views: RefCell>>, +} + +impl VulkanSurface { + /// Creates a Skia Vulkan rendering surface from the given Vulkano device, queue family index, surface, + /// and size. + pub fn from_surface( + physical_device: Arc, + queue_family_index: u32, + surface: Arc, + size: PhysicalWindowSize, + ) -> Result { + /* + eprintln!( + "Vulkan device: {} (type: {:?})", + physical_device.properties().device_name, + physical_device.properties().device_type, + );*/ + + let (device, mut queues) = Device::new( + physical_device.clone(), + DeviceCreateInfo { + enabled_extensions: DeviceExtensions { + khr_swapchain: true, + ..DeviceExtensions::empty() + }, + queue_create_infos: vec![QueueCreateInfo { + queue_family_index, + ..Default::default() + }], + ..Default::default() + }, + ) + .map_err(|dev_err| format!("Failed to create suitable logical Vulkan device: {dev_err}"))?; + let queue = queues.next().ok_or_else(|| "Not Vulkan device queue found".to_string())?; + + let (swapchain, swapchain_images) = { + let surface_capabilities = device + .physical_device() + .surface_capabilities(&surface, Default::default()) + .map_err(|vke| format!("Error matching Vulkan surface capabilities: {vke}"))?; + let image_format = vulkano::format::Format::B8G8R8A8_UNORM; + + Swapchain::new( + device.clone(), + surface.clone(), + SwapchainCreateInfo { + min_image_count: surface_capabilities.min_image_count, + image_format, + image_extent: [size.width, size.height], + image_usage: ImageUsage::COLOR_ATTACHMENT, + composite_alpha: surface_capabilities + .supported_composite_alpha + .into_iter() + .next() + .ok_or_else(|| { + "fatal: Vulkan surface capabilities missing composite alpha descriptor" + .to_string() + })?, + ..Default::default() + }, + ) + .map_err(|vke| format!("Error creating Vulkan swapchain: {vke}"))? + }; + + let mut swapchain_image_views = Vec::with_capacity(swapchain_images.len()); + + for image in &swapchain_images { + swapchain_image_views.push(ImageView::new_default(image.clone()).map_err(|vke| { + format!("fatal: Error creating image view for swap chain image: {vke}") + })?); + } + + let instance = physical_device.instance(); + let library = instance.library(); + + let get_proc = |of| unsafe { + let result = match of { + skia_safe::gpu::vk::GetProcOf::Instance(instance, name) => { + library.get_instance_proc_addr(ash::vk::Instance::from_raw(instance as _), name) + } + skia_safe::gpu::vk::GetProcOf::Device(device, name) => { + (instance.fns().v1_0.get_device_proc_addr)( + ash::vk::Device::from_raw(device as _), + name, + ) + } + }; + + match result { + Some(f) => f as _, + None => { + //println!("resolve of {} failed", of.name().to_str().unwrap()); + core::ptr::null() + } + } + }; + + // Cap the Vulkan API version Skia uses. Skia otherwise assumes the highest version reported + // by the physical device and tries to load functions and request features for it. That fails + // on drivers where the logical device we created only negotiated a lower version, so limit it + // to the version vulkano negotiated for this physical device under our instance. + let api_version = physical_device.api_version(); + let max_api_version = skia_safe::gpu::vk::Version::new( + api_version.major as _, + api_version.minor as _, + api_version.patch as _, + ); + + let backend_context = unsafe { + skia_safe::gpu::vk::BackendContext::new_builder( + instance.handle().as_raw() as _, + physical_device.handle().as_raw() as _, + device.handle().as_raw() as _, + (queue.handle().as_raw() as _, queue.queue_index() as _), + &get_proc, + Some(max_api_version), + ) + .build() + }; + + let gr_context = skia_safe::gpu::direct_contexts::make_vulkan(&backend_context, None) + .ok_or_else(|| { + format!( + "Error creating Skia Vulkan context (max api version {}.{}.{})", + api_version.major, api_version.minor, api_version.patch + ) + })?; + + let previous_frame_end = RefCell::new(Some(sync::now(device.clone()).boxed())); + + Ok(Self { + gr_context: RefCell::new(gr_context), + recreate_swapchain: Cell::new(false), + device, + previous_frame_end, + queue, + swapchain: RefCell::new(swapchain), + swapchain_images: RefCell::new(swapchain_images), + swapchain_image_views: RefCell::new(swapchain_image_views), + }) + } + + /// Returns a clone of the shared swapchain. + pub fn swapchain(&self) -> Arc { + self.swapchain.borrow().clone() + } +} + +impl super::Surface for VulkanSurface { + fn new( + shared_context: &SkiaSharedContext, + window_handle: Arc, + display_handle: Arc, + size: PhysicalWindowSize, + requested_graphics_api: Option, + ) -> Result { + if requested_graphics_api.is_some_and(|api| !matches!(api, RequestedGraphicsAPI::Vulkan)) { + return Err("Requested non-Vulkan rendering with Vulkan renderer".into()); + } + + let instance = shared_context.0.shared_vulkan_context()?.instance.clone(); + + let window_handle = window_handle + .window_handle() + .map_err(|e| format!("error obtaining window handle for skia vulkan renderer: {e}"))?; + let display_handle = display_handle + .display_handle() + .map_err(|e| format!("error obtaining display handle for skia vulkan renderer: {e}"))?; + + let surface = create_surface(&instance, window_handle, display_handle) + .map_err(|surface_err| format!("Error creating Vulkan surface: {surface_err}"))?; + + let device_extensions = + DeviceExtensions { khr_swapchain: true, ..DeviceExtensions::empty() }; + let (physical_device, queue_family_index) = instance + .enumerate_physical_devices() + .map_err(|vke| format!("Error enumerating physical Vulkan devices: {vke}"))? + .filter(|p| p.supported_extensions().contains(&device_extensions)) + .filter_map(|p| { + p.queue_family_properties() + .iter() + .enumerate() + .position(|(i, q)| { + q.queue_flags.intersects(QueueFlags::GRAPHICS) + && p.surface_support(i as u32, &surface).unwrap_or(false) + }) + .map(|i| (p, i as u32)) + }) + .min_by_key(|(p, _)| match p.properties().device_type { + PhysicalDeviceType::DiscreteGpu => 0, + PhysicalDeviceType::IntegratedGpu => 1, + PhysicalDeviceType::VirtualGpu => 2, + PhysicalDeviceType::Cpu => 3, + PhysicalDeviceType::Other => 4, + _ => 5, + }) + .ok_or_else(|| "Vulkan: Failed to find suitable physical device".to_string())?; + + Self::from_surface(physical_device, queue_family_index, surface, size) + } + + fn name(&self) -> &'static str { + "vulkan" + } + + fn resize_event( + &self, + _size: PhysicalWindowSize, + ) -> Result<(), i_slint_core::platform::PlatformError> { + self.recreate_swapchain.set(true); + Ok(()) + } + + fn render( + &self, + _window: &Window, + size: PhysicalWindowSize, + callback: &dyn Fn( + &skia_safe::Canvas, + Option<&mut skia_safe::gpu::DirectContext>, + u8, + ) -> Option, + pre_present_callback: &RefCell>>, + ) -> Result { + let gr_context = &mut self.gr_context.borrow_mut(); + + let device = self.device.clone(); + + self.previous_frame_end.borrow_mut().as_mut().unwrap().cleanup_finished(); + + if self.recreate_swapchain.take() { + let mut swapchain = self.swapchain.borrow_mut(); + let (new_swapchain, new_images) = swapchain + .recreate(SwapchainCreateInfo { + image_extent: [size.width, size.height], + ..swapchain.create_info() + }) + .map_err(|vke| format!("Error re-creating Vulkan swap chain: {vke}"))?; + + *swapchain = new_swapchain; + + let mut new_swapchain_image_views = Vec::with_capacity(new_images.len()); + + for image in &new_images { + new_swapchain_image_views.push(ImageView::new_default(image.clone()).map_err( + |vke| format!("fatal: Error creating image view for swap chain image: {vke}"), + )?); + } + + *self.swapchain_images.borrow_mut() = new_images; + *self.swapchain_image_views.borrow_mut() = new_swapchain_image_views; + } + + let swapchain = self.swapchain.borrow().clone(); + + #[cfg_attr(slint_nightly_test, allow(non_exhaustive_omitted_patterns))] + let (image_index, suboptimal, acquire_future) = + match vulkano::swapchain::acquire_next_image(swapchain.clone(), None) + .map_err(Validated::unwrap) + { + Ok(r) => r, + Err(VulkanError::OutOfDate) => { + self.recreate_swapchain.set(true); + return Ok(DrawOutcome::Occluded); // Try again next frame + } + Err(e) => return Err(format!("Vulkan: failed to acquire next image: {e}").into()), + }; + + if suboptimal { + self.recreate_swapchain.set(true); + } + + let width = swapchain.image_extent()[0]; + let width: i32 = width + .try_into() + .map_err(|_| format!("internal error: invalid swapchain image width {width}"))?; + let height = swapchain.image_extent()[1]; + let height: i32 = height + .try_into() + .map_err(|_| format!("internal error: invalid swapchain image height {height}"))?; + + let image_view = self.swapchain_image_views.borrow()[image_index as usize].clone(); + let image_object = image_view.image(); + + let format = image_view.format(); + + debug_assert_eq!(format, vulkano::format::Format::B8G8R8A8_UNORM); + let (vk_format, color_type) = + (skia_safe::gpu::vk::Format::B8G8R8A8_UNORM, skia_safe::ColorType::BGRA8888); + + let alloc = skia_safe::gpu::vk::Alloc::default(); + let image_info = &unsafe { + skia_safe::gpu::vk::ImageInfo::new( + image_object.handle().as_raw() as _, + alloc, + skia_safe::gpu::vk::ImageTiling::OPTIMAL, + skia_safe::gpu::vk::ImageLayout::COLOR_ATTACHMENT_OPTIMAL, + vk_format, + 1, + None, + None, + None, + None, + ) + }; + + let render_target = + &skia_safe::gpu::backend_render_targets::make_vk((width, height), image_info); + + let mut skia_surface = skia_safe::gpu::surfaces::wrap_backend_render_target( + gr_context, + render_target, + skia_safe::gpu::SurfaceOrigin::TopLeft, + color_type, + None, + None, + ) + .ok_or_else(|| "Error creating Skia Vulkan surface".to_string())?; + + callback(skia_surface.canvas(), Some(gr_context), 0); + + drop(skia_surface); + + gr_context.submit(None); + + if let Some(pre_present_callback) = pre_present_callback.borrow_mut().as_mut() { + pre_present_callback(); + } + + let future = self + .previous_frame_end + .borrow_mut() + .take() + .unwrap() + .join(acquire_future) + .then_swapchain_present( + self.queue.clone(), + SwapchainPresentInfo::swapchain_image_index(swapchain.clone(), image_index), + ) + .then_signal_fence_and_flush(); + + #[cfg_attr(slint_nightly_test, allow(non_exhaustive_omitted_patterns))] + match future.map_err(Validated::unwrap) { + Ok(future) => { + *self.previous_frame_end.borrow_mut() = Some(future.boxed()); + } + Err(VulkanError::OutOfDate) => { + self.recreate_swapchain.set(true); + *self.previous_frame_end.borrow_mut() = Some(sync::now(device.clone()).boxed()); + } + Err(e) => { + *self.previous_frame_end.borrow_mut() = Some(sync::now(device.clone()).boxed()); + return Err(format!("Skia Vulkan renderer: failed to flush future: {e}").into()); + } + } + + Ok(DrawOutcome::Success) + } + + fn bits_per_pixel(&self) -> Result { + #[cfg_attr(slint_nightly_test, allow(non_exhaustive_omitted_patterns))] + Ok(match self.swapchain.borrow().image_format() { + vulkano::format::Format::B8G8R8A8_UNORM => 32, + fmt => { + return Err(format!( + "Skia Vulkan Renderer: Unsupported swapchain image format found {fmt:?}" + ) + .into()); + } + }) + } + + fn as_any(&self) -> &dyn core::any::Any { + self + } +} + +// FIXME(madsmtm): Why are we doing this instead of using `Surface::from_window`? +fn create_surface( + instance: &Arc, + window_handle: raw_window_handle::WindowHandle<'_>, + display_handle: raw_window_handle::DisplayHandle<'_>, +) -> Result, vulkano::Validated> { + #[cfg_attr(slint_nightly_test, allow(non_exhaustive_omitted_patterns))] + match (window_handle.as_raw(), display_handle.as_raw()) { + #[cfg(target_vendor = "apple")] + (raw_window_handle::RawWindowHandle::AppKit(handle), _) => unsafe { + let layer = raw_window_metal::Layer::from_ns_view(handle.ns_view); + Surface::from_metal(instance.clone(), layer.as_ptr().as_ptr(), None) + }, + #[cfg(target_vendor = "apple")] + (raw_window_handle::RawWindowHandle::UiKit(handle), _) => unsafe { + let layer = raw_window_metal::Layer::from_ui_view(handle.ui_view); + Surface::from_metal(instance.clone(), layer.as_ptr().as_ptr(), None) + }, + ( + raw_window_handle::RawWindowHandle::Xlib(raw_window_handle::XlibWindowHandle { + window, + .. + }), + raw_window_handle::RawDisplayHandle::Xlib(display), + ) => unsafe { + Surface::from_xlib(instance.clone(), display.display.unwrap().as_ptr(), window, None) + }, + ( + raw_window_handle::RawWindowHandle::Xcb(raw_window_handle::XcbWindowHandle { + window, + .. + }), + raw_window_handle::RawDisplayHandle::Xcb(raw_window_handle::XcbDisplayHandle { + connection, + .. + }), + ) => unsafe { + Surface::from_xcb(instance.clone(), connection.unwrap().as_ptr(), window.get(), None) + }, + ( + raw_window_handle::RawWindowHandle::Wayland(raw_window_handle::WaylandWindowHandle { + surface, + .. + }), + raw_window_handle::RawDisplayHandle::Wayland(raw_window_handle::WaylandDisplayHandle { + display, + .. + }), + ) => unsafe { + Surface::from_wayland(instance.clone(), display.as_ptr(), surface.as_ptr(), None) + }, + ( + raw_window_handle::RawWindowHandle::Win32(raw_window_handle::Win32WindowHandle { + hwnd, + hinstance, + .. + }), + _, + ) => unsafe { + Surface::from_win32(instance.clone(), hinstance.unwrap().get(), hwnd.get(), None) + }, + _ => unimplemented!(), + } +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/wgpu_28_surface.rs b/third_party/i-slint-renderer-skia-1.17.1/wgpu_28_surface.rs new file mode 100644 index 0000000..89a954b --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/wgpu_28_surface.rs @@ -0,0 +1,381 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +#[cfg(feature = "unstable-wgpu-28")] +use i_slint_core::api::GraphicsAPI; +use i_slint_core::api::{PhysicalSize as PhysicalWindowSize, Window}; +use i_slint_core::graphics::RequestedGraphicsAPI; +use i_slint_core::partial_renderer::DirtyRegion; +use i_slint_core::platform::PlatformError; +use i_slint_core::renderer::DrawOutcome; + +use std::cell::RefCell; +use std::sync::Arc; + +use wgpu_28 as wgpu; + +use crate::SkiaSharedContext; + +#[cfg(target_family = "windows")] +mod dx12; +#[cfg(target_vendor = "apple")] +mod metal; +#[cfg(all(target_family = "unix", not(target_vendor = "apple")))] +mod vulkan; + +/// Skia rendering surface backed by WGPU. Supports both on-screen rendering (with a +/// window surface) and offscreen rendering into caller-provided textures. +pub struct WGPUSurface { + pub(crate) gr_context: RefCell, + instance: wgpu::Instance, + device: wgpu::Device, + queue: wgpu::Queue, + surface_config: RefCell>, + surface: Option>, + textures_to_transition_for_sampling: RefCell>, + pub(crate) backend: Backend, +} + +impl WGPUSurface { + pub fn new_with_surface( + surface_target: impl Into, + size: PhysicalWindowSize, + requested_graphics_api: Option, + ) -> Result { + let (instance, adapter, device, queue, surface) = + i_slint_core::graphics::wgpu_28::init_instance_adapter_device_queue_surface( + surface_target, + requested_graphics_api, + wgpu::Backends::GL /* we're not mapping that to skia because we can't save/restore state */ + .union(if cfg!(target_os = "windows") { + wgpu::Backends::VULKAN + } else { + wgpu::Backends::empty() + }), + )?; + + let mut surface_config = + surface.get_default_config(&adapter, size.width, size.height).unwrap(); + + let swapchain_capabilities = surface.get_capabilities(&adapter); + let swapchain_format = swapchain_capabilities + .formats + .iter() + .find(|f| { + matches!(f, wgpu::TextureFormat::Rgba8Unorm | wgpu::TextureFormat::Bgra8Unorm) + }) + .copied() + .unwrap_or_else(|| swapchain_capabilities.formats[0]); + surface_config.format = swapchain_format; + surface.configure(&device, &surface_config); + + let backend: Backend = adapter.get_info().backend.try_into()?; + + let gr_context = backend.make_context(&adapter, &device, &queue); + + Ok(Self { + gr_context: RefCell::new( + gr_context.ok_or_else(|| { + PlatformError::from("Failed to create Skia context from WGPU") + })?, + ), + instance, + device, + queue, + surface_config: Some(surface_config).into(), + surface: Some(surface), + textures_to_transition_for_sampling: RefCell::new(Vec::new()), + backend, + }) + } + + // Only used by SkiaWGPURenderer, which is gated on wgpu-29 — the wgpu-28 + // path doesn't currently expose an offscreen renderer publicly. + #[allow(dead_code)] + pub(crate) fn new_offscreen( + instance: wgpu::Instance, + device: wgpu::Device, + queue: wgpu::Queue, + backend: Backend, + gr_context: skia_safe::gpu::DirectContext, + ) -> Self { + Self { + gr_context: RefCell::new(gr_context), + instance, + device, + queue, + surface_config: None.into(), + surface: None, + textures_to_transition_for_sampling: RefCell::new(Vec::new()), + backend, + } + } + + /// Transitions any imported wgpu textures to sampling state and flushes + /// the Skia graphics context. Must be called after rendering to ensure + /// Skia's GPU work is submitted. + pub(crate) fn flush_and_submit(&self, gr_context: &mut skia_safe::gpu::DirectContext) { + let textures_to_transition = self.textures_to_transition_for_sampling.take(); + if !textures_to_transition.is_empty() { + let mut encoder = self.device.create_command_encoder(&wgpu::CommandEncoderDescriptor { + label: Some("Skia texture transition encoder"), + }); + encoder.transition_resources( + std::iter::empty(), + textures_to_transition.iter().map(|texture| wgpu::TextureTransition { + texture, + selector: None, + state: wgpu::TextureUses::RESOURCE, + }), + ); + self.queue.submit(Some(encoder.finish())); + } + + gr_context.submit(None); + } +} + +impl crate::Surface for WGPUSurface { + fn new( + _shared_context: &SkiaSharedContext, + window_handle: Arc, + display_handle: Arc, + size: PhysicalWindowSize, + requested_graphics_api: Option, + ) -> Result { + Self::new_with_surface( + Box::new(WindowAndDisplayHandle(window_handle, display_handle)) + as Box, + size, + requested_graphics_api, + ) + } + + fn name(&self) -> &'static str { + if self.surface.is_some() { "wgpu" } else { "wgpu-texture" } + } + + fn resize_event(&self, size: PhysicalWindowSize) -> Result<(), PlatformError> { + let mut surface_config_opt = self.surface_config.borrow_mut(); + let (Some(surface_config), Some(surface)) = (surface_config_opt.as_mut(), &self.surface) + else { + return Ok(()); + }; + + // Skip reconfigure if size hasn't changed — DRM/KMS surfaces don't + // support being reconfigured. + if surface_config.width == size.width && surface_config.height == size.height { + return Ok(()); + } + + { + let gr_context = &mut self.gr_context.borrow_mut(); + // This is brute force, but for the lack of access to the fences this seems to work: Avoid any pending work so that + // IDXGISwapChain::ResizeBuffers doesn't complain that the surface is still in use. + gr_context.flush_submit_and_sync_cpu(); + } + + // Prefer FIFO modes over possible Mailbox setting for frame pacing and better energy efficiency. + surface_config.present_mode = wgpu::PresentMode::AutoVsync; + surface_config.width = size.width; + surface_config.height = size.height; + + surface.configure(&self.device, surface_config); + Ok(()) + } + + fn render( + &self, + _window: &Window, + _size: PhysicalWindowSize, + callback: &dyn Fn( + &skia_safe::Canvas, + Option<&mut skia_safe::gpu::DirectContext>, + u8, + ) -> Option, + pre_present_callback: &RefCell>>, + ) -> Result { + let (Some(surface), Some(surface_config)) = (&self.surface, &*self.surface_config.borrow()) + else { + return Err("WGPUSurface::render() called on offscreen surface".into()); + }; + + let gr_context = &mut self.gr_context.borrow_mut(); + + let frame = match surface.get_current_texture() { + Ok(texture) => texture, + Err(wgpu::SurfaceError::Timeout) => return Ok(DrawOutcome::Timeout), + // Outdated or lost: re-configure and try once. If the surface still + // doesn't yield a texture, treat it as occluded so the caller re-arms. + Err(_) => { + surface.configure(&self.device, surface_config); + match surface.get_current_texture() { + Ok(texture) => texture, + Err(_) => return Ok(DrawOutcome::Occluded), + } + } + }; + + let skia_surface = self.backend.make_surface(gr_context, &frame.texture); + + let mut skia_surface = skia_surface + .ok_or_else(|| PlatformError::from("Failed to create Skia surface from WGPU"))?; + + callback(skia_surface.canvas(), Some(gr_context), 0); + + self.flush_and_submit(gr_context); + + if let Some(pre_present_callback) = pre_present_callback.borrow_mut().as_mut() { + pre_present_callback(); + } + + frame.present(); + + Ok(DrawOutcome::Success) + } + + fn bits_per_pixel(&self) -> Result { + if let Some(surface_config) = &*self.surface_config.borrow() { + Ok(match surface_config.format { + wgpu_28::TextureFormat::Rgba8Unorm + | wgpu_28::TextureFormat::Rgba8UnormSrgb + | wgpu_28::TextureFormat::Bgra8Unorm + | wgpu_28::TextureFormat::Bgra8UnormSrgb => 32, + fmt => return Err(format!("Unsupported surface format {:#?}", fmt).into()), + }) + } else { + // All supported render-target formats (Rgba8Unorm, Bgra8Unorm, and sRGB variants) are 32bpp. + Ok(32) + } + } + + #[cfg(feature = "unstable-wgpu-28")] + fn with_graphics_api(&self, callback: &mut dyn FnMut(GraphicsAPI<'_>)) { + let api = i_slint_core::graphics::create_graphics_api_wgpu_28( + self.instance.clone(), + self.device.clone(), + self.queue.clone(), + ); + callback(api) + } + + #[cfg(any(feature = "unstable-wgpu-28", feature = "unstable-wgpu-29"))] + fn import_wgpu_texture( + &self, + canvas: &skia_safe::Canvas, + any_wgpu_texture: &i_slint_core::graphics::WGPUTexture, + ) -> Option { + let texture = match any_wgpu_texture { + #[cfg(feature = "unstable-wgpu-28")] + i_slint_core::graphics::WGPUTexture::WGPU28Texture(texture) => texture.clone(), + #[cfg(feature = "unstable-wgpu-29")] + i_slint_core::graphics::WGPUTexture::WGPU29Texture(..) => return None, + }; + + // Skia won't submit commands right away, so remember the texture and transition before + // submitting. + self.textures_to_transition_for_sampling.borrow_mut().push(texture.clone()); + + self.backend.import_texture(canvas, texture) + } +} + +struct WindowAndDisplayHandle( + Arc, + Arc, +); + +impl raw_window_handle::HasWindowHandle for WindowAndDisplayHandle { + fn window_handle( + &self, + ) -> Result, raw_window_handle::HandleError> { + self.0.window_handle() + } +} + +impl raw_window_handle::HasDisplayHandle for WindowAndDisplayHandle { + fn display_handle( + &self, + ) -> Result, raw_window_handle::HandleError> { + self.1.display_handle() + } +} + +pub(crate) enum Backend { + #[cfg(target_vendor = "apple")] + Metal, + #[cfg(target_family = "windows")] + Dx12, + #[cfg(all(target_family = "unix", not(target_vendor = "apple")))] + Vulkan, +} + +impl TryFrom for Backend { + type Error = PlatformError; + + fn try_from(wgpu_backend: wgpu::Backend) -> Result { + match wgpu_backend { + wgpu_28::Backend::Noop => { + Err(PlatformError::from("Cannot use WGPU Noop backend with Skia")) + } + #[cfg(all(target_family = "unix", not(target_vendor = "apple")))] + wgpu_28::Backend::Vulkan => Ok(Self::Vulkan), + #[cfg(target_vendor = "apple")] + wgpu_28::Backend::Metal => Ok(Self::Metal), + #[cfg(target_family = "windows")] + wgpu_28::Backend::Dx12 => Ok(Self::Dx12), + other => Err(PlatformError::from(format!( + "Unsupported WGPU backend for use with Skia: {}", + other + ))), + } + } +} + +impl Backend { + pub(crate) fn make_context( + &self, + _adapter: &wgpu::Adapter, + device: &wgpu::Device, + queue: &wgpu::Queue, + ) -> Option { + match self { + #[cfg(target_vendor = "apple")] + Self::Metal => metal::make_metal_context(device, queue), + #[cfg(target_family = "windows")] + Self::Dx12 => unsafe { dx12::make_dx12_context(&_adapter, &device, &queue) }, + #[cfg(all(target_family = "unix", not(target_vendor = "apple")))] + Self::Vulkan => unsafe { vulkan::make_vulkan_context(device, queue) }, + } + } + + pub(crate) fn make_surface( + &self, + gr_context: &mut skia_safe::gpu::DirectContext, + texture: &wgpu::Texture, + ) -> Option { + match self { + #[cfg(target_vendor = "apple")] + Self::Metal => unsafe { metal::make_metal_surface(gr_context, texture) }, + #[cfg(target_family = "windows")] + Self::Dx12 => unsafe { dx12::make_dx12_surface(gr_context, texture) }, + #[cfg(all(target_family = "unix", not(target_vendor = "apple")))] + Self::Vulkan => unsafe { vulkan::make_vulkan_surface(gr_context, texture) }, + } + } + + pub(crate) fn import_texture( + &self, + canvas: &skia_safe::Canvas, + texture: wgpu::Texture, + ) -> Option { + match self { + #[cfg(target_vendor = "apple")] + Self::Metal => unsafe { metal::import_metal_texture(canvas, texture) }, + #[cfg(target_family = "windows")] + Self::Dx12 => unsafe { dx12::import_dx12_texture(canvas, texture) }, + #[cfg(all(target_family = "unix", not(target_vendor = "apple")))] + Self::Vulkan => unsafe { vulkan::import_vulkan_texture(canvas, texture) }, + } + } +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/wgpu_28_surface/dx12.rs b/third_party/i-slint-renderer-skia-1.17.1/wgpu_28_surface/dx12.rs new file mode 100644 index 0000000..e0e9af2 --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/wgpu_28_surface/dx12.rs @@ -0,0 +1,173 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +use windows::Win32::Graphics::Direct3D12::{D3D12_RESOURCE_STATE_PRESENT, ID3D12Resource}; +use windows::Win32::Graphics::Dxgi::Common::DXGI_STANDARD_MULTISAMPLE_QUALITY_PATTERN; +use windows::Win32::Graphics::Dxgi::Common::{ + DXGI_FORMAT_B8G8R8A8_UNORM, DXGI_FORMAT_R8G8B8A8_UNORM, DXGI_FORMAT_R8G8B8A8_UNORM_SRGB, +}; + +use wgpu_28 as wgpu; + +/// # Safety +/// `resource` must be a valid D3D12 resource for the lifetime of the returned Surface. +unsafe fn wrap_dx12_texture( + width: i32, + height: i32, + gr_context: &mut skia_safe::gpu::DirectContext, + resource: ID3D12Resource, + dxgi_format: windows::Win32::Graphics::Dxgi::Common::DXGI_FORMAT, + color_type: skia_safe::ColorType, +) -> Option { + unsafe { + let texture_info = skia_safe::gpu::d3d::TextureResourceInfo { + resource, + alloc: None, + resource_state: D3D12_RESOURCE_STATE_PRESENT, + format: dxgi_format, + sample_count: 1, + level_count: 1, + sample_quality_pattern: DXGI_STANDARD_MULTISAMPLE_QUALITY_PATTERN, + protected: skia_safe::gpu::Protected::No, + }; + let backend_render_target = + skia_safe::gpu::BackendRenderTarget::new_d3d((width, height), &texture_info); + skia_safe::gpu::surfaces::wrap_backend_render_target( + gr_context, + &backend_render_target, + skia_safe::gpu::SurfaceOrigin::TopLeft, + color_type, + None, + None, + ) + } +} + +/// # Safety +/// The caller must ensure `texture` was created by a DX12-backed wgpu device and remains +/// valid for the lifetime of the returned `skia_safe::Surface`. +pub unsafe fn make_dx12_surface( + gr_context: &mut skia_safe::gpu::DirectContext, + texture: &wgpu::Texture, +) -> Option { + // SAFETY: texture is borrowed for the duration of this call; the D3D12 resource is + // ref-counted (COM) and cloned into Skia's internal BackendRenderTarget via wrap_dx12_texture. + unsafe { + let dx12_texture = texture.as_hal::()?; + let resource = windows_core::Interface::from_raw(windows_core::Interface::into_raw( + dx12_texture.raw_resource().clone(), + )); + let size = texture.size(); + let (dxgi_format, color_type) = match texture.format() { + wgpu::TextureFormat::Rgba8Unorm => { + (DXGI_FORMAT_R8G8B8A8_UNORM, skia_safe::ColorType::RGBA8888) + } + wgpu::TextureFormat::Rgba8UnormSrgb => { + (DXGI_FORMAT_R8G8B8A8_UNORM_SRGB, skia_safe::ColorType::SRGBA8888) + } + wgpu::TextureFormat::Bgra8Unorm => { + (DXGI_FORMAT_B8G8R8A8_UNORM, skia_safe::ColorType::BGRA8888) + } + _ => return None, + }; + wrap_dx12_texture( + size.width as i32, + size.height as i32, + gr_context, + resource, + dxgi_format, + color_type, + ) + } +} + +#[allow(non_snake_case)] +pub unsafe fn import_dx12_texture( + canvas: &skia_safe::Canvas, + texture: wgpu::Texture, +) -> Option { + unsafe { + let dx12_texture = texture.as_hal::(); + + let resource: ID3D12Resource = windows_core::Interface::from_raw( + windows_core::Interface::into_raw(dx12_texture.unwrap().raw_resource().clone()), + ); + + let dxgi_texture_format = resource.GetDesc().Format; + + let color_type = match dxgi_texture_format { + DXGI_FORMAT_R8G8B8A8_UNORM => skia_safe::ColorType::RGBA8888, + DXGI_FORMAT_R8G8B8A8_UNORM_SRGB => skia_safe::ColorType::SRGBA8888, + DXGI_FORMAT_B8G8R8A8_UNORM => skia_safe::ColorType::BGRA8888, + _ => return None, + }; + + let texture_info = skia_safe::gpu::d3d::TextureResourceInfo { + resource, + alloc: None, + resource_state: D3D12_RESOURCE_STATE_PRESENT, + format: dxgi_texture_format, + sample_count: 1, + level_count: 1, + sample_quality_pattern: DXGI_STANDARD_MULTISAMPLE_QUALITY_PATTERN, + protected: skia_safe::gpu::Protected::No, + }; + let size = texture.size(); + + let backend_texture = skia_safe::gpu::BackendTexture::new_d3d( + (size.width as i32, size.height as i32), + &texture_info, + ); + + Some( + skia_safe::image::Image::from_texture( + canvas.recording_context().as_mut().unwrap(), + &backend_texture, + skia_safe::gpu::SurfaceOrigin::TopLeft, + color_type, + skia_safe::AlphaType::Unpremul, + None, + ) + .unwrap(), + ) + } +} + +pub unsafe fn make_dx12_context( + adapter: &wgpu::Adapter, + _device: &wgpu::Device, + queue: &wgpu::Queue, +) -> Option { + let backend = unsafe { + let maybe_dx12_queue = queue.as_hal::(); + let dx12_adapter = adapter.as_hal::().unwrap(); + + maybe_dx12_queue.map(|dx12_queue| { + let dx12_queue_raw = dx12_queue.as_raw(); + let mut dx12_device_old: Option = + None; + dx12_queue_raw.GetDevice(&mut dx12_device_old as _).unwrap(); + let dx12_device_old = dx12_device_old.unwrap(); + let dx12_device = windows_core::Interface::from_raw(windows_core::Interface::into_raw( + dx12_device_old, + )); + + let idxgiadapter_3: windows::Win32::Graphics::Dxgi::IDXGIAdapter3 = + dx12_adapter.as_raw().clone().into(); + + skia_safe::gpu::d3d::BackendContext { + adapter: windows_core::Interface::from_raw(windows_core::Interface::into_raw( + idxgiadapter_3, + )), + device: dx12_device, + queue: windows_core::Interface::from_raw(windows_core::Interface::into_raw( + dx12_queue_raw.clone(), + )), + memory_allocator: None, + protected_context: skia_safe::gpu::Protected::No, + } + }) + }; + + skia_safe::gpu::DirectContext::new_d3d(&backend.unwrap(), None) +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/wgpu_28_surface/metal.rs b/third_party/i-slint-renderer-skia-1.17.1/wgpu_28_surface/metal.rs new file mode 100644 index 0000000..02267e0 --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/wgpu_28_surface/metal.rs @@ -0,0 +1,114 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +use foreign_types::ForeignType; + +use skia_safe::gpu::mtl; + +use wgpu_28 as wgpu; + +/// # Safety +/// `metal_handle` must be a valid Metal texture handle for the lifetime of the returned Surface. +unsafe fn wrap_metal_texture( + width: i32, + height: i32, + gr_context: &mut skia_safe::gpu::DirectContext, + metal_handle: mtl::Handle, + color_type: skia_safe::ColorType, +) -> Option { + unsafe { + let texture_info = mtl::TextureInfo::new(metal_handle); + let backend_render_target = + skia_safe::gpu::backend_render_targets::make_mtl((width, height), &texture_info); + skia_safe::gpu::surfaces::wrap_backend_render_target( + gr_context, + &backend_render_target, + skia_safe::gpu::SurfaceOrigin::TopLeft, + color_type, + None, + None, + ) + } +} + +/// # Safety +/// The caller must ensure `texture` was created by a Metal-backed wgpu device and remains +/// valid for the lifetime of the returned `skia_safe::Surface`. +pub unsafe fn make_metal_surface( + gr_context: &mut skia_safe::gpu::DirectContext, + texture: &wgpu::Texture, +) -> Option { + // SAFETY: texture is borrowed for the duration of this call; the Metal handle is copied + // into Skia's internal BackendRenderTarget via wrap_metal_texture. + unsafe { + let metal_texture = texture.as_hal::()?; + let handle = metal_texture.raw_handle().as_ptr() as mtl::Handle; + let size = texture.size(); + let color_type = match texture.format() { + wgpu::TextureFormat::Bgra8Unorm => skia_safe::ColorType::BGRA8888, + wgpu::TextureFormat::Rgba8Unorm => skia_safe::ColorType::RGBA8888, + wgpu::TextureFormat::Rgba8UnormSrgb => skia_safe::ColorType::SRGBA8888, + _ => return None, + }; + wrap_metal_texture(size.width as i32, size.height as i32, gr_context, handle, color_type) + } +} + +pub unsafe fn import_metal_texture( + canvas: &skia_safe::Canvas, + texture: wgpu::Texture, +) -> Option { + unsafe { + let metal_texture = texture.as_hal::(); + + let texture_info = + mtl::TextureInfo::new(metal_texture.unwrap().raw_handle().as_ptr() as mtl::Handle); + let size = texture.size(); + + let backend_texture = skia_safe::gpu::backend_textures::make_mtl( + (size.width as _, size.height as _), + skia_safe::gpu::Mipmapped::No, + &texture_info, + "Borrowed Metal texture", + ); + Some( + skia_safe::image::Image::from_texture( + canvas.recording_context().as_mut().unwrap(), + &backend_texture, + skia_safe::gpu::SurfaceOrigin::TopLeft, + match texture.format() { + wgpu::TextureFormat::Rgba8Unorm => skia_safe::ColorType::RGBA8888, + wgpu::TextureFormat::Rgba8UnormSrgb => skia_safe::ColorType::SRGBA8888, + _ => return None, + }, + skia_safe::AlphaType::Unpremul, + None, + ) + .unwrap(), + ) + } +} + +pub fn make_metal_context( + device: &wgpu::Device, + queue: &wgpu::Queue, +) -> Option { + let backend = unsafe { + let maybe_metal_device = device.as_hal::(); + let maybe_metal_queue = queue.as_hal::(); + + maybe_metal_device.and_then(|metal_device| { + let metal_device_raw = metal_device.raw_device(); + + maybe_metal_queue.map(|metal_queue| { + let metal_queue_raw = &*metal_queue.as_raw().lock(); + mtl::BackendContext::new( + metal_device_raw.as_ptr() as mtl::Handle, + metal_queue_raw.as_ptr() as mtl::Handle, + ) + }) + })? + }; + + skia_safe::gpu::direct_contexts::make_metal(&backend, None) +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/wgpu_28_surface/vulkan.rs b/third_party/i-slint-renderer-skia-1.17.1/wgpu_28_surface/vulkan.rs new file mode 100644 index 0000000..7334544 --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/wgpu_28_surface/vulkan.rs @@ -0,0 +1,188 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +use ash::vk::Handle; +use skia_safe::gpu::vk; + +use wgpu_28 as wgpu; + +fn vk_format_and_color_type( + format: wgpu::TextureFormat, +) -> Option<(skia_safe::gpu::vk::Format, skia_safe::ColorType)> { + match format { + wgpu::TextureFormat::Rgba8Unorm => { + Some((skia_safe::gpu::vk::Format::R8G8B8A8_UNORM, skia_safe::ColorType::RGBA8888)) + } + wgpu::TextureFormat::Rgba8UnormSrgb => { + Some((skia_safe::gpu::vk::Format::R8G8B8A8_SRGB, skia_safe::ColorType::SRGBA8888)) + } + wgpu::TextureFormat::Bgra8Unorm => { + Some((skia_safe::gpu::vk::Format::B8G8R8A8_UNORM, skia_safe::ColorType::BGRA8888)) + } + _ => None, + } +} + +/// # Safety +/// `vk_image_raw` must be a valid Vulkan image handle for the lifetime of the returned Surface. +unsafe fn wrap_vulkan_texture( + width: i32, + height: i32, + gr_context: &mut skia_safe::gpu::DirectContext, + vk_image_raw: u64, + vk_format: skia_safe::gpu::vk::Format, + color_type: skia_safe::ColorType, +) -> Option { + unsafe { + let texture_info = &skia_safe::gpu::vk::ImageInfo::new( + vk_image_raw as _, + skia_safe::gpu::vk::Alloc::default(), + skia_safe::gpu::vk::ImageTiling::OPTIMAL, + skia_safe::gpu::vk::ImageLayout::COLOR_ATTACHMENT_OPTIMAL, + vk_format, + 1, + None, + None, + None, + None, + ); + let backend_render_target = + skia_safe::gpu::backend_render_targets::make_vk((width, height), texture_info); + skia_safe::gpu::surfaces::wrap_backend_render_target( + gr_context, + &backend_render_target, + skia_safe::gpu::SurfaceOrigin::TopLeft, + color_type, + None, + None, + ) + } +} + +/// # Safety +/// The caller must ensure `texture` was created by a Vulkan-backed wgpu device and remains +/// valid for the lifetime of the returned `skia_safe::Surface`. +pub unsafe fn make_vulkan_surface( + gr_context: &mut skia_safe::gpu::DirectContext, + texture: &wgpu::Texture, +) -> Option { + // SAFETY: texture is borrowed for the duration of this call; the Vulkan handle is copied + // into Skia's internal BackendRenderTarget via wrap_vulkan_texture. + unsafe { + let vulkan_texture = texture.as_hal::()?; + let vk_image_raw = vulkan_texture.raw_handle().as_raw(); + let size = texture.size(); + let (vk_format, color_type) = vk_format_and_color_type(texture.format())?; + wrap_vulkan_texture( + size.width as i32, + size.height as i32, + gr_context, + vk_image_raw, + vk_format, + color_type, + ) + } +} + +pub unsafe fn import_vulkan_texture( + canvas: &skia_safe::Canvas, + texture: wgpu::Texture, +) -> Option { + unsafe { + let vulkan_texture = texture.as_hal::(); + + let alloc = skia_safe::gpu::vk::Alloc::default(); + + let (vk_format, color_type) = match texture.format() { + wgpu::TextureFormat::Rgba8Unorm => { + (skia_safe::gpu::vk::Format::R8G8B8A8_UNORM, skia_safe::ColorType::RGBA8888) + } + wgpu::TextureFormat::Rgba8UnormSrgb => { + (skia_safe::gpu::vk::Format::R8G8B8A8_SRGB, skia_safe::ColorType::SRGBA8888) + } + wgpu::TextureFormat::Bgra8Unorm => { + (skia_safe::gpu::vk::Format::B8G8R8A8_UNORM, skia_safe::ColorType::BGRA8888) + } + _ => return None, + }; + + let texture_info = &skia_safe::gpu::vk::ImageInfo::new( + vulkan_texture.unwrap().raw_handle().as_raw() as _, + alloc, + skia_safe::gpu::vk::ImageTiling::OPTIMAL, + skia_safe::gpu::vk::ImageLayout::COLOR_ATTACHMENT_OPTIMAL, + vk_format, + 1, + None, + None, + None, + None, + ); + + let size = texture.size(); + + let backend_texture = skia_safe::gpu::backend_textures::make_vk( + (size.width as _, size.height as _), + texture_info, + "Borrowed Vulkan texture", + ); + Some( + skia_safe::image::Image::from_texture( + canvas.recording_context().as_mut().unwrap(), + &backend_texture, + skia_safe::gpu::SurfaceOrigin::TopLeft, + color_type, + skia_safe::AlphaType::Unpremul, + None, + ) + .unwrap(), + ) + } +} + +pub unsafe fn make_vulkan_context( + device: &wgpu::Device, + queue: &wgpu::Queue, +) -> Option { + unsafe { + let vulkan_device = device.as_hal::()?; + let vulkan_queue = queue.as_hal::()?; + + let vulkan_queue_raw = vulkan_queue.as_raw(); + + let get_proc = |of| { + let result = match of { + skia_safe::gpu::vk::GetProcOf::Instance(instance, name) => vulkan_device + .shared_instance() + .entry() + .get_instance_proc_addr(ash::vk::Instance::from_raw(instance as _), name), + skia_safe::gpu::vk::GetProcOf::Device(device, name) => vulkan_device + .shared_instance() + .raw_instance() + .get_device_proc_addr(ash::vk::Device::from_raw(device as _), name), + }; + + match result { + Some(f) => f as _, + None => { + //println!("resolve of {} failed", of.name().to_str().unwrap()); + core::ptr::null() + } + } + }; + + // WGPU 28 is locked to vulkan 1.3 and skia assumes the highest vulkan API version of the + // physical device is chosen, causing it to ask for unsupported features/functions. + let backend = vk::BackendContext::new_builder( + vulkan_device.shared_instance().raw_instance().handle().as_raw() as _, + vulkan_device.raw_physical_device().as_raw() as _, + vulkan_device.raw_device().handle().as_raw() as _, + (vulkan_queue_raw.as_raw() as _, vulkan_device.queue_family_index() as _), + &get_proc, + Some(vk::Version::new(1, 3, 0)), + ) + .build(); + + skia_safe::gpu::direct_contexts::make_vulkan(&backend, None) + } +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/wgpu_29_surface.rs b/third_party/i-slint-renderer-skia-1.17.1/wgpu_29_surface.rs new file mode 100644 index 0000000..03aea4b --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/wgpu_29_surface.rs @@ -0,0 +1,388 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +#[cfg(feature = "unstable-wgpu-29")] +use i_slint_core::api::GraphicsAPI; +use i_slint_core::api::{PhysicalSize as PhysicalWindowSize, Window}; +use i_slint_core::graphics::RequestedGraphicsAPI; +use i_slint_core::partial_renderer::DirtyRegion; +use i_slint_core::platform::PlatformError; +use i_slint_core::renderer::DrawOutcome; + +use std::cell::RefCell; +use std::sync::Arc; + +use wgpu_29 as wgpu; + +use crate::SkiaSharedContext; + +#[cfg(target_family = "windows")] +mod dx12; +#[cfg(target_vendor = "apple")] +mod metal; +#[cfg(all(target_family = "unix", not(target_vendor = "apple")))] +mod vulkan; + +/// Skia rendering surface backed by WGPU. Supports both on-screen rendering (with a +/// window surface) and offscreen rendering into caller-provided textures. +pub struct WGPUSurface { + pub(crate) gr_context: RefCell, + instance: wgpu::Instance, + device: wgpu::Device, + queue: wgpu::Queue, + surface_config: RefCell>, + surface: Option>, + textures_to_transition_for_sampling: RefCell>, + pub(crate) backend: Backend, +} + +impl WGPUSurface { + pub fn new_with_surface( + surface_target: impl Into, + size: PhysicalWindowSize, + requested_graphics_api: Option, + ) -> Result { + let (instance, adapter, device, queue, surface) = + i_slint_core::graphics::wgpu_29::init_instance_adapter_device_queue_surface( + surface_target, + requested_graphics_api, + wgpu::Backends::GL /* we're not mapping that to skia because we can't save/restore state */ + .union(if cfg!(target_os = "windows") { + wgpu::Backends::VULKAN + } else { + wgpu::Backends::empty() + }), + )?; + + let mut surface_config = + surface.get_default_config(&adapter, size.width, size.height).unwrap(); + + let swapchain_capabilities = surface.get_capabilities(&adapter); + let swapchain_format = swapchain_capabilities + .formats + .iter() + .find(|f| { + matches!(f, wgpu::TextureFormat::Rgba8Unorm | wgpu::TextureFormat::Bgra8Unorm) + }) + .copied() + .unwrap_or_else(|| swapchain_capabilities.formats[0]); + surface_config.format = swapchain_format; + surface.configure(&device, &surface_config); + + let backend: Backend = adapter.get_info().backend.try_into()?; + + let gr_context = backend.make_context(&adapter, &device, &queue); + + Ok(Self { + gr_context: RefCell::new( + gr_context.ok_or_else(|| { + PlatformError::from("Failed to create Skia context from WGPU") + })?, + ), + instance, + device, + queue, + surface_config: Some(surface_config).into(), + surface: Some(surface), + textures_to_transition_for_sampling: RefCell::new(Vec::new()), + backend, + }) + } + + pub(crate) fn new_offscreen( + instance: wgpu::Instance, + device: wgpu::Device, + queue: wgpu::Queue, + backend: Backend, + gr_context: skia_safe::gpu::DirectContext, + ) -> Self { + Self { + gr_context: RefCell::new(gr_context), + instance, + device, + queue, + surface_config: None.into(), + surface: None, + textures_to_transition_for_sampling: RefCell::new(Vec::new()), + backend, + } + } + + /// Transitions any imported wgpu textures to sampling state and flushes + /// the Skia graphics context. Must be called after rendering to ensure + /// Skia's GPU work is submitted. + pub(crate) fn flush_and_submit(&self, gr_context: &mut skia_safe::gpu::DirectContext) { + let textures_to_transition = self.textures_to_transition_for_sampling.take(); + if !textures_to_transition.is_empty() { + let mut encoder = self.device.create_command_encoder(&wgpu::CommandEncoderDescriptor { + label: Some("Skia texture transition encoder"), + }); + encoder.transition_resources( + std::iter::empty(), + textures_to_transition.iter().map(|texture| wgpu::TextureTransition { + texture, + selector: None, + state: wgpu::TextureUses::RESOURCE, + }), + ); + self.queue.submit(Some(encoder.finish())); + } + + gr_context.submit(None); + } +} + +impl crate::Surface for WGPUSurface { + fn new( + _shared_context: &SkiaSharedContext, + window_handle: Arc, + display_handle: Arc, + size: PhysicalWindowSize, + requested_graphics_api: Option, + ) -> Result { + Self::new_with_surface( + Box::new(WindowAndDisplayHandle(window_handle, display_handle)) + as Box, + size, + requested_graphics_api, + ) + } + + fn name(&self) -> &'static str { + if self.surface.is_some() { "wgpu" } else { "wgpu-texture" } + } + + fn resize_event(&self, size: PhysicalWindowSize) -> Result<(), PlatformError> { + let mut surface_config_opt = self.surface_config.borrow_mut(); + let (Some(surface_config), Some(surface)) = (surface_config_opt.as_mut(), &self.surface) + else { + return Ok(()); + }; + + // Skip reconfigure if size hasn't changed — DRM/KMS surfaces don't + // support being reconfigured. + if surface_config.width == size.width && surface_config.height == size.height { + return Ok(()); + } + + { + let gr_context = &mut self.gr_context.borrow_mut(); + // This is brute force, but for the lack of access to the fences this seems to work: Avoid any pending work so that + // IDXGISwapChain::ResizeBuffers doesn't complain that the surface is still in use. + gr_context.flush_submit_and_sync_cpu(); + } + + // Prefer FIFO modes over possible Mailbox setting for frame pacing and better energy efficiency. + surface_config.present_mode = wgpu::PresentMode::AutoVsync; + surface_config.width = size.width; + surface_config.height = size.height; + + surface.configure(&self.device, surface_config); + Ok(()) + } + + fn render( + &self, + _window: &Window, + _size: PhysicalWindowSize, + callback: &dyn Fn( + &skia_safe::Canvas, + Option<&mut skia_safe::gpu::DirectContext>, + u8, + ) -> Option, + pre_present_callback: &RefCell>>, + ) -> Result { + let (Some(surface), Some(surface_config)) = (&self.surface, &*self.surface_config.borrow()) + else { + return Err("WGPUSurface::render() called on offscreen surface".into()); + }; + + let gr_context = &mut self.gr_context.borrow_mut(); + + let frame = match surface.get_current_texture() { + wgpu::CurrentSurfaceTexture::Success(t) => t, + wgpu::CurrentSurfaceTexture::Occluded => return Ok(DrawOutcome::Occluded), + wgpu::CurrentSurfaceTexture::Timeout => return Ok(DrawOutcome::Timeout), + wgpu::CurrentSurfaceTexture::Validation => { + return Err("WGPU surface validation error in get_current_texture".into()); + } + stale @ (wgpu::CurrentSurfaceTexture::Outdated + | wgpu::CurrentSurfaceTexture::Suboptimal(_) + | wgpu::CurrentSurfaceTexture::Lost) => { + // `Suboptimal` carries a live `SurfaceTexture`; matched with `_` it is not bound, + // so the value returned by `get_current_texture()` keeps it alive across the + // `surface.configure()` below — which wgpu forbids ("`SurfaceOutput` must be + // dropped before a new `Surface` is made"), panicking on the first frame on + // Wayland. Drop it first. (`Outdated`/`Lost` carry nothing → no-op.) + drop(stale); + surface.configure(&self.device, surface_config); + match surface.get_current_texture() { + wgpu::CurrentSurfaceTexture::Success(t) => t, + _ => return Ok(DrawOutcome::Occluded), + } + } + }; + + let skia_surface = self.backend.make_surface(gr_context, &frame.texture); + + let mut skia_surface = skia_surface + .ok_or_else(|| PlatformError::from("Failed to create Skia surface from WGPU"))?; + + callback(skia_surface.canvas(), Some(gr_context), 0); + + self.flush_and_submit(gr_context); + + if let Some(pre_present_callback) = pre_present_callback.borrow_mut().as_mut() { + pre_present_callback(); + } + + frame.present(); + + Ok(DrawOutcome::Success) + } + + fn bits_per_pixel(&self) -> Result { + if let Some(surface_config) = &*self.surface_config.borrow() { + Ok(match surface_config.format { + wgpu_29::TextureFormat::Rgba8Unorm + | wgpu_29::TextureFormat::Rgba8UnormSrgb + | wgpu_29::TextureFormat::Bgra8Unorm + | wgpu_29::TextureFormat::Bgra8UnormSrgb => 32, + fmt => return Err(format!("Unsupported surface format {:#?}", fmt).into()), + }) + } else { + // All supported render-target formats (Rgba8Unorm, Bgra8Unorm, and sRGB variants) are 32bpp. + Ok(32) + } + } + + #[cfg(feature = "unstable-wgpu-29")] + fn with_graphics_api(&self, callback: &mut dyn FnMut(GraphicsAPI<'_>)) { + let api = i_slint_core::graphics::create_graphics_api_wgpu_29( + self.instance.clone(), + self.device.clone(), + self.queue.clone(), + ); + callback(api) + } + + #[cfg(any(feature = "unstable-wgpu-28", feature = "unstable-wgpu-29"))] + fn import_wgpu_texture( + &self, + canvas: &skia_safe::Canvas, + any_wgpu_texture: &i_slint_core::graphics::WGPUTexture, + ) -> Option { + let texture = match any_wgpu_texture { + #[cfg(feature = "unstable-wgpu-28")] + i_slint_core::graphics::WGPUTexture::WGPU28Texture(..) => return None, + #[cfg(feature = "unstable-wgpu-29")] + i_slint_core::graphics::WGPUTexture::WGPU29Texture(texture) => texture.clone(), + }; + + // Skia won't submit commands right away, so remember the texture and transition before + // submitting. + self.textures_to_transition_for_sampling.borrow_mut().push(texture.clone()); + + self.backend.import_texture(canvas, texture) + } +} + +struct WindowAndDisplayHandle( + Arc, + Arc, +); + +impl raw_window_handle::HasWindowHandle for WindowAndDisplayHandle { + fn window_handle( + &self, + ) -> Result, raw_window_handle::HandleError> { + self.0.window_handle() + } +} + +impl raw_window_handle::HasDisplayHandle for WindowAndDisplayHandle { + fn display_handle( + &self, + ) -> Result, raw_window_handle::HandleError> { + self.1.display_handle() + } +} + +pub(crate) enum Backend { + #[cfg(target_vendor = "apple")] + Metal, + #[cfg(target_family = "windows")] + Dx12, + #[cfg(all(target_family = "unix", not(target_vendor = "apple")))] + Vulkan, +} + +impl TryFrom for Backend { + type Error = PlatformError; + + fn try_from(wgpu_backend: wgpu::Backend) -> Result { + match wgpu_backend { + wgpu_29::Backend::Noop => { + Err(PlatformError::from("Cannot use WGPU Noop backend with Skia")) + } + #[cfg(all(target_family = "unix", not(target_vendor = "apple")))] + wgpu_29::Backend::Vulkan => Ok(Self::Vulkan), + #[cfg(target_vendor = "apple")] + wgpu_29::Backend::Metal => Ok(Self::Metal), + #[cfg(target_family = "windows")] + wgpu_29::Backend::Dx12 => Ok(Self::Dx12), + other => Err(PlatformError::from(format!( + "Unsupported WGPU backend for use with Skia: {}", + other + ))), + } + } +} + +impl Backend { + pub(crate) fn make_context( + &self, + _adapter: &wgpu::Adapter, + device: &wgpu::Device, + queue: &wgpu::Queue, + ) -> Option { + match self { + #[cfg(target_vendor = "apple")] + Self::Metal => metal::make_metal_context(device, queue), + #[cfg(target_family = "windows")] + Self::Dx12 => unsafe { dx12::make_dx12_context(&_adapter, &device, &queue) }, + #[cfg(all(target_family = "unix", not(target_vendor = "apple")))] + Self::Vulkan => unsafe { vulkan::make_vulkan_context(device, queue) }, + } + } + + pub(crate) fn make_surface( + &self, + gr_context: &mut skia_safe::gpu::DirectContext, + texture: &wgpu::Texture, + ) -> Option { + match self { + #[cfg(target_vendor = "apple")] + Self::Metal => unsafe { metal::make_metal_surface(gr_context, texture) }, + #[cfg(target_family = "windows")] + Self::Dx12 => unsafe { dx12::make_dx12_surface(gr_context, texture) }, + #[cfg(all(target_family = "unix", not(target_vendor = "apple")))] + Self::Vulkan => unsafe { vulkan::make_vulkan_surface(gr_context, texture) }, + } + } + + pub(crate) fn import_texture( + &self, + canvas: &skia_safe::Canvas, + texture: wgpu::Texture, + ) -> Option { + match self { + #[cfg(target_vendor = "apple")] + Self::Metal => unsafe { metal::import_metal_texture(canvas, texture) }, + #[cfg(target_family = "windows")] + Self::Dx12 => unsafe { dx12::import_dx12_texture(canvas, texture) }, + #[cfg(all(target_family = "unix", not(target_vendor = "apple")))] + Self::Vulkan => unsafe { vulkan::import_vulkan_texture(canvas, texture) }, + } + } +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/wgpu_29_surface/dx12.rs b/third_party/i-slint-renderer-skia-1.17.1/wgpu_29_surface/dx12.rs new file mode 100644 index 0000000..d75375a --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/wgpu_29_surface/dx12.rs @@ -0,0 +1,173 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +use windows::Win32::Graphics::Direct3D12::{D3D12_RESOURCE_STATE_PRESENT, ID3D12Resource}; +use windows::Win32::Graphics::Dxgi::Common::DXGI_STANDARD_MULTISAMPLE_QUALITY_PATTERN; +use windows::Win32::Graphics::Dxgi::Common::{ + DXGI_FORMAT_B8G8R8A8_UNORM, DXGI_FORMAT_R8G8B8A8_UNORM, DXGI_FORMAT_R8G8B8A8_UNORM_SRGB, +}; + +use wgpu_29 as wgpu; + +/// # Safety +/// `resource` must be a valid D3D12 resource for the lifetime of the returned Surface. +unsafe fn wrap_dx12_texture( + width: i32, + height: i32, + gr_context: &mut skia_safe::gpu::DirectContext, + resource: ID3D12Resource, + dxgi_format: windows::Win32::Graphics::Dxgi::Common::DXGI_FORMAT, + color_type: skia_safe::ColorType, +) -> Option { + unsafe { + let texture_info = skia_safe::gpu::d3d::TextureResourceInfo { + resource, + alloc: None, + resource_state: D3D12_RESOURCE_STATE_PRESENT, + format: dxgi_format, + sample_count: 1, + level_count: 1, + sample_quality_pattern: DXGI_STANDARD_MULTISAMPLE_QUALITY_PATTERN, + protected: skia_safe::gpu::Protected::No, + }; + let backend_render_target = + skia_safe::gpu::BackendRenderTarget::new_d3d((width, height), &texture_info); + skia_safe::gpu::surfaces::wrap_backend_render_target( + gr_context, + &backend_render_target, + skia_safe::gpu::SurfaceOrigin::TopLeft, + color_type, + None, + None, + ) + } +} + +/// # Safety +/// The caller must ensure `texture` was created by a DX12-backed wgpu device and remains +/// valid for the lifetime of the returned `skia_safe::Surface`. +pub unsafe fn make_dx12_surface( + gr_context: &mut skia_safe::gpu::DirectContext, + texture: &wgpu::Texture, +) -> Option { + // SAFETY: texture is borrowed for the duration of this call; the D3D12 resource is + // ref-counted (COM) and cloned into Skia's internal BackendRenderTarget via wrap_dx12_texture. + unsafe { + let dx12_texture = texture.as_hal::()?; + let resource = windows_core::Interface::from_raw(windows_core::Interface::into_raw( + dx12_texture.raw_resource().clone(), + )); + let size = texture.size(); + let (dxgi_format, color_type) = match texture.format() { + wgpu::TextureFormat::Rgba8Unorm => { + (DXGI_FORMAT_R8G8B8A8_UNORM, skia_safe::ColorType::RGBA8888) + } + wgpu::TextureFormat::Rgba8UnormSrgb => { + (DXGI_FORMAT_R8G8B8A8_UNORM_SRGB, skia_safe::ColorType::SRGBA8888) + } + wgpu::TextureFormat::Bgra8Unorm => { + (DXGI_FORMAT_B8G8R8A8_UNORM, skia_safe::ColorType::BGRA8888) + } + _ => return None, + }; + wrap_dx12_texture( + size.width as i32, + size.height as i32, + gr_context, + resource, + dxgi_format, + color_type, + ) + } +} + +#[allow(non_snake_case)] +pub unsafe fn import_dx12_texture( + canvas: &skia_safe::Canvas, + texture: wgpu::Texture, +) -> Option { + unsafe { + let dx12_texture = texture.as_hal::(); + + let resource: ID3D12Resource = windows_core::Interface::from_raw( + windows_core::Interface::into_raw(dx12_texture.unwrap().raw_resource().clone()), + ); + + let dxgi_texture_format = resource.GetDesc().Format; + + let color_type = match dxgi_texture_format { + DXGI_FORMAT_R8G8B8A8_UNORM => skia_safe::ColorType::RGBA8888, + DXGI_FORMAT_R8G8B8A8_UNORM_SRGB => skia_safe::ColorType::SRGBA8888, + DXGI_FORMAT_B8G8R8A8_UNORM => skia_safe::ColorType::BGRA8888, + _ => return None, + }; + + let texture_info = skia_safe::gpu::d3d::TextureResourceInfo { + resource, + alloc: None, + resource_state: D3D12_RESOURCE_STATE_PRESENT, + format: dxgi_texture_format, + sample_count: 1, + level_count: 1, + sample_quality_pattern: DXGI_STANDARD_MULTISAMPLE_QUALITY_PATTERN, + protected: skia_safe::gpu::Protected::No, + }; + let size = texture.size(); + + let backend_texture = skia_safe::gpu::BackendTexture::new_d3d( + (size.width as i32, size.height as i32), + &texture_info, + ); + + Some( + skia_safe::image::Image::from_texture( + canvas.recording_context().as_mut().unwrap(), + &backend_texture, + skia_safe::gpu::SurfaceOrigin::TopLeft, + color_type, + skia_safe::AlphaType::Unpremul, + None, + ) + .unwrap(), + ) + } +} + +pub unsafe fn make_dx12_context( + adapter: &wgpu::Adapter, + _device: &wgpu::Device, + queue: &wgpu::Queue, +) -> Option { + let backend = unsafe { + let maybe_dx12_queue = queue.as_hal::(); + let dx12_adapter = adapter.as_hal::().unwrap(); + + maybe_dx12_queue.map(|dx12_queue| { + let dx12_queue_raw = dx12_queue.as_raw(); + let mut dx12_device_old: Option = + None; + dx12_queue_raw.GetDevice(&mut dx12_device_old as _).unwrap(); + let dx12_device_old = dx12_device_old.unwrap(); + let dx12_device = windows_core::Interface::from_raw(windows_core::Interface::into_raw( + dx12_device_old, + )); + + let idxgiadapter_3: windows::Win32::Graphics::Dxgi::IDXGIAdapter3 = + dx12_adapter.as_raw().clone().into(); + + skia_safe::gpu::d3d::BackendContext { + adapter: windows_core::Interface::from_raw(windows_core::Interface::into_raw( + idxgiadapter_3, + )), + device: dx12_device, + queue: windows_core::Interface::from_raw(windows_core::Interface::into_raw( + dx12_queue_raw.clone(), + )), + memory_allocator: None, + protected_context: skia_safe::gpu::Protected::No, + } + }) + }; + + skia_safe::gpu::DirectContext::new_d3d(&backend.unwrap(), None) +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/wgpu_29_surface/metal.rs b/third_party/i-slint-renderer-skia-1.17.1/wgpu_29_surface/metal.rs new file mode 100644 index 0000000..5c319e7 --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/wgpu_29_surface/metal.rs @@ -0,0 +1,120 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +use objc2::rc::Retained; +use objc2::runtime::ProtocolObject; +use objc2_metal::{MTLCommandQueue, MTLTexture}; +use skia_safe::gpu::mtl; + +use wgpu_29 as wgpu; + +/// # Safety +/// `metal_handle` must be a valid Metal texture handle for the lifetime of the returned Surface. +unsafe fn wrap_metal_texture( + width: i32, + height: i32, + gr_context: &mut skia_safe::gpu::DirectContext, + metal_handle: mtl::Handle, + color_type: skia_safe::ColorType, +) -> Option { + unsafe { + let texture_info = mtl::TextureInfo::new(metal_handle); + let backend_render_target = + skia_safe::gpu::backend_render_targets::make_mtl((width, height), &texture_info); + skia_safe::gpu::surfaces::wrap_backend_render_target( + gr_context, + &backend_render_target, + skia_safe::gpu::SurfaceOrigin::TopLeft, + color_type, + None, + None, + ) + } +} + +/// # Safety +/// The caller must ensure `texture` was created by a Metal-backed wgpu device and remains +/// valid for the lifetime of the returned `skia_safe::Surface`. +pub unsafe fn make_metal_surface( + gr_context: &mut skia_safe::gpu::DirectContext, + texture: &wgpu::Texture, +) -> Option { + // SAFETY: texture is borrowed for the duration of this call; the Metal handle is copied + // into Skia's internal BackendRenderTarget via wrap_metal_texture. + unsafe { + let metal_texture = texture.as_hal::()?; + let handle = + metal_texture.raw_handle() as *const ProtocolObject as mtl::Handle; + let size = texture.size(); + let color_type = match texture.format() { + wgpu::TextureFormat::Bgra8Unorm => skia_safe::ColorType::BGRA8888, + wgpu::TextureFormat::Rgba8Unorm => skia_safe::ColorType::RGBA8888, + wgpu::TextureFormat::Rgba8UnormSrgb => skia_safe::ColorType::SRGBA8888, + _ => return None, + }; + wrap_metal_texture(size.width as i32, size.height as i32, gr_context, handle, color_type) + } +} + +pub unsafe fn import_metal_texture( + canvas: &skia_safe::Canvas, + texture: wgpu::Texture, +) -> Option { + unsafe { + let metal_texture = texture.as_hal::(); + + let texture_info = mtl::TextureInfo::new(metal_texture.unwrap().raw_handle() + as *const ProtocolObject + as mtl::Handle); + let size = texture.size(); + + let backend_texture = skia_safe::gpu::backend_textures::make_mtl( + (size.width as _, size.height as _), + skia_safe::gpu::Mipmapped::No, + &texture_info, + "Borrowed Metal texture", + ); + Some( + skia_safe::image::Image::from_texture( + canvas.recording_context().as_mut().unwrap(), + &backend_texture, + skia_safe::gpu::SurfaceOrigin::TopLeft, + match texture.format() { + wgpu::TextureFormat::Rgba8Unorm => skia_safe::ColorType::RGBA8888, + wgpu::TextureFormat::Rgba8UnormSrgb => skia_safe::ColorType::SRGBA8888, + _ => return None, + }, + skia_safe::AlphaType::Unpremul, + None, + ) + .unwrap(), + ) + } +} + +pub fn make_metal_context( + device: &wgpu::Device, + queue: &wgpu::Queue, +) -> Option { + let backend = unsafe { + let maybe_metal_device = device.as_hal::(); + let maybe_metal_queue = queue.as_hal::(); + + maybe_metal_device.and_then(|metal_device| { + let metal_device_raw = metal_device.raw_device(); + + maybe_metal_queue.map(|metal_queue| { + // Share wgpu's command queue with Skia, so that Metal's per-queue hazard tracking + // orders wgpu's texture writes before Skia samples them. + let metal_queue_raw: *const ProtocolObject = + metal_queue.as_raw(); + mtl::BackendContext::new( + Retained::as_ptr(metal_device_raw) as mtl::Handle, + metal_queue_raw as mtl::Handle, + ) + }) + })? + }; + + skia_safe::gpu::direct_contexts::make_metal(&backend, None) +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/wgpu_29_surface/vulkan.rs b/third_party/i-slint-renderer-skia-1.17.1/wgpu_29_surface/vulkan.rs new file mode 100644 index 0000000..feaddcd --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/wgpu_29_surface/vulkan.rs @@ -0,0 +1,188 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +use ash::vk::Handle; +use skia_safe::gpu::vk; + +use wgpu_29 as wgpu; + +fn vk_format_and_color_type( + format: wgpu::TextureFormat, +) -> Option<(skia_safe::gpu::vk::Format, skia_safe::ColorType)> { + match format { + wgpu::TextureFormat::Rgba8Unorm => { + Some((skia_safe::gpu::vk::Format::R8G8B8A8_UNORM, skia_safe::ColorType::RGBA8888)) + } + wgpu::TextureFormat::Rgba8UnormSrgb => { + Some((skia_safe::gpu::vk::Format::R8G8B8A8_SRGB, skia_safe::ColorType::SRGBA8888)) + } + wgpu::TextureFormat::Bgra8Unorm => { + Some((skia_safe::gpu::vk::Format::B8G8R8A8_UNORM, skia_safe::ColorType::BGRA8888)) + } + _ => None, + } +} + +/// # Safety +/// `vk_image_raw` must be a valid Vulkan image handle for the lifetime of the returned Surface. +unsafe fn wrap_vulkan_texture( + width: i32, + height: i32, + gr_context: &mut skia_safe::gpu::DirectContext, + vk_image_raw: u64, + vk_format: skia_safe::gpu::vk::Format, + color_type: skia_safe::ColorType, +) -> Option { + unsafe { + let texture_info = &skia_safe::gpu::vk::ImageInfo::new( + vk_image_raw as _, + skia_safe::gpu::vk::Alloc::default(), + skia_safe::gpu::vk::ImageTiling::OPTIMAL, + skia_safe::gpu::vk::ImageLayout::COLOR_ATTACHMENT_OPTIMAL, + vk_format, + 1, + None, + None, + None, + None, + ); + let backend_render_target = + skia_safe::gpu::backend_render_targets::make_vk((width, height), texture_info); + skia_safe::gpu::surfaces::wrap_backend_render_target( + gr_context, + &backend_render_target, + skia_safe::gpu::SurfaceOrigin::TopLeft, + color_type, + None, + None, + ) + } +} + +/// # Safety +/// The caller must ensure `texture` was created by a Vulkan-backed wgpu device and remains +/// valid for the lifetime of the returned `skia_safe::Surface`. +pub unsafe fn make_vulkan_surface( + gr_context: &mut skia_safe::gpu::DirectContext, + texture: &wgpu::Texture, +) -> Option { + // SAFETY: texture is borrowed for the duration of this call; the Vulkan handle is copied + // into Skia's internal BackendRenderTarget via wrap_vulkan_texture. + unsafe { + let vulkan_texture = texture.as_hal::()?; + let vk_image_raw = vulkan_texture.raw_handle().as_raw(); + let size = texture.size(); + let (vk_format, color_type) = vk_format_and_color_type(texture.format())?; + wrap_vulkan_texture( + size.width as i32, + size.height as i32, + gr_context, + vk_image_raw, + vk_format, + color_type, + ) + } +} + +pub unsafe fn import_vulkan_texture( + canvas: &skia_safe::Canvas, + texture: wgpu::Texture, +) -> Option { + unsafe { + let vulkan_texture = texture.as_hal::(); + + let alloc = skia_safe::gpu::vk::Alloc::default(); + + let (vk_format, color_type) = match texture.format() { + wgpu::TextureFormat::Rgba8Unorm => { + (skia_safe::gpu::vk::Format::R8G8B8A8_UNORM, skia_safe::ColorType::RGBA8888) + } + wgpu::TextureFormat::Rgba8UnormSrgb => { + (skia_safe::gpu::vk::Format::R8G8B8A8_SRGB, skia_safe::ColorType::SRGBA8888) + } + wgpu::TextureFormat::Bgra8Unorm => { + (skia_safe::gpu::vk::Format::B8G8R8A8_UNORM, skia_safe::ColorType::BGRA8888) + } + _ => return None, + }; + + let texture_info = &skia_safe::gpu::vk::ImageInfo::new( + vulkan_texture.unwrap().raw_handle().as_raw() as _, + alloc, + skia_safe::gpu::vk::ImageTiling::OPTIMAL, + skia_safe::gpu::vk::ImageLayout::COLOR_ATTACHMENT_OPTIMAL, + vk_format, + 1, + None, + None, + None, + None, + ); + + let size = texture.size(); + + let backend_texture = skia_safe::gpu::backend_textures::make_vk( + (size.width as _, size.height as _), + texture_info, + "Borrowed Vulkan texture", + ); + Some( + skia_safe::image::Image::from_texture( + canvas.recording_context().as_mut().unwrap(), + &backend_texture, + skia_safe::gpu::SurfaceOrigin::TopLeft, + color_type, + skia_safe::AlphaType::Unpremul, + None, + ) + .unwrap(), + ) + } +} + +pub unsafe fn make_vulkan_context( + device: &wgpu::Device, + queue: &wgpu::Queue, +) -> Option { + unsafe { + let vulkan_device = device.as_hal::()?; + let vulkan_queue = queue.as_hal::()?; + + let vulkan_queue_raw = vulkan_queue.as_raw(); + + let get_proc = |of| { + let result = match of { + skia_safe::gpu::vk::GetProcOf::Instance(instance, name) => vulkan_device + .shared_instance() + .entry() + .get_instance_proc_addr(ash::vk::Instance::from_raw(instance as _), name), + skia_safe::gpu::vk::GetProcOf::Device(device, name) => vulkan_device + .shared_instance() + .raw_instance() + .get_device_proc_addr(ash::vk::Device::from_raw(device as _), name), + }; + + match result { + Some(f) => f as _, + None => { + //println!("resolve of {} failed", of.name().to_str().unwrap()); + core::ptr::null() + } + } + }; + + // WGPU 29 is locked to vulkan 1.3 and skia assumes the highest vulkan API version of the + // physical device is chosen, causing it to ask for unsupported features/functions. + let backend = vk::BackendContext::new_builder( + vulkan_device.shared_instance().raw_instance().handle().as_raw() as _, + vulkan_device.raw_physical_device().as_raw() as _, + vulkan_device.raw_device().handle().as_raw() as _, + (vulkan_queue_raw.as_raw() as _, vulkan_device.queue_family_index() as _), + &get_proc, + Some(vk::Version::new(1, 3, 0)), + ) + .build(); + + skia_safe::gpu::direct_contexts::make_vulkan(&backend, None) + } +} diff --git a/third_party/i-slint-renderer-skia-1.17.1/wgpu_renderer.rs b/third_party/i-slint-renderer-skia-1.17.1/wgpu_renderer.rs new file mode 100644 index 0000000..9ba703b --- /dev/null +++ b/third_party/i-slint-renderer-skia-1.17.1/wgpu_renderer.rs @@ -0,0 +1,198 @@ +// Copyright © SixtyFPS GmbH +// SPDX-License-Identifier: GPL-3.0-only OR LicenseRef-Slint-Royalty-free-2.0 OR LicenseRef-Slint-Software-3.0 + +// cSpell: ignore rasterizers +use std::pin::Pin; +use std::rc::Rc; + +use i_slint_core::platform::PlatformError; +use i_slint_core::renderer::RendererSealed; +use i_slint_core::window::WindowAdapter; + +use wgpu_29 as wgpu; + +use crate::wgpu_29_surface::{Backend, WGPUSurface}; +use crate::{SkiaRenderer, SkiaSharedContext}; + +/// Use the Skia renderer with WGPU when implementing a custom Slint platform where you want the +/// scene to be rendered into a WGPU texture. The rendering is done using the +/// [Skia](https://skia.org/) library with platform-native GPU acceleration. +/// +/// This is the Skia equivalent of `FemtoVGWGPURenderer`, offering superior font rendering +/// quality through platform-native text rasterizers. +/// +/// Rendering notifier callbacks registered via +/// [`Window::set_rendering_notifier()`](i_slint_core::api::Window::set_rendering_notifier) +/// will receive [`GraphicsAPI::WGPU29`](i_slint_core::api::GraphicsAPI::WGPU29) with the +/// renderer's instance, device, and queue. +pub struct SkiaWGPURenderer { + renderer: SkiaRenderer, + surface: WGPUSurface, +} + +impl SkiaWGPURenderer { + /// Creates a new SkiaWGPURenderer. + /// + /// The `instance`, `adapter`, `device` and `queue` are the WGPU resources used for rendering. + /// The `adapter` is needed to determine the GPU backend and create the Skia graphics context. + /// + /// The wgpu resources are also provided to rendering notifier callbacks via + /// [`GraphicsAPI::WGPU29`](i_slint_core::api::GraphicsAPI::WGPU29). + pub fn new( + instance: wgpu::Instance, + adapter: wgpu::Adapter, + device: wgpu::Device, + queue: wgpu::Queue, + ) -> Result { + let backend: Backend = adapter.get_info().backend.try_into()?; + + let gr_context = backend.make_context(&adapter, &device, &queue).ok_or_else(|| { + PlatformError::from("Failed to create Skia graphics context from WGPU") + })?; + + let surface = WGPUSurface::new_offscreen(instance, device, queue, backend, gr_context); + + let shared_context = SkiaSharedContext::default(); + // Use SkiaRenderer::default() to stay resilient to field additions, then disable + // partial rendering — there is no buffer age tracking for external texture targets. + let mut renderer = SkiaRenderer::default(&shared_context); + renderer.partial_rendering_state = None; + + Ok(Self { renderer, surface }) + } + + /// Render the scene to the given texture. + /// + /// The texture must have been created with `RENDER_ATTACHMENT` usage and have a supported + /// format. Supported formats depend on the GPU backend: `Rgba8Unorm` and `Rgba8UnormSrgb` + /// are supported on all backends; `Bgra8Unorm` is additionally supported on Metal and Vulkan. + pub fn render_to_texture(&self, texture: &wgpu::Texture) -> Result<(), PlatformError> { + self.renderer.invoke_rendering_notifier_setup(&self.surface)?; + + let gr_context = &mut self.surface.gr_context.borrow_mut(); + + let mut skia_surface = + self.surface.backend.make_surface(gr_context, texture).ok_or_else(|| { + PlatformError::from("Failed to wrap WGPU texture as Skia render target") + })?; + + let window_adapter = self.renderer.window_adapter()?; + let window = window_adapter.window(); + + self.renderer.render_to_canvas( + skia_surface.canvas(), + 0., + (0., 0.), + Some(gr_context), + 0, + Some(&self.surface), + window, + None, + ); + + self.surface.flush_and_submit(gr_context); + + Ok(()) + } +} + +#[doc(hidden)] +impl RendererSealed for SkiaWGPURenderer { + fn text_size( + &self, + text_item: Pin<&dyn i_slint_core::item_rendering::RenderString>, + item_rc: &i_slint_core::items::ItemRc, + max_width: Option, + text_wrap: i_slint_core::items::TextWrap, + ) -> i_slint_core::lengths::LogicalSize { + self.renderer.text_size(text_item, item_rc, max_width, text_wrap) + } + + fn char_size( + &self, + text_item: Pin<&dyn i_slint_core::item_rendering::HasFont>, + item_rc: &i_slint_core::items::ItemRc, + ch: char, + ) -> i_slint_core::lengths::LogicalSize { + self.renderer.char_size(text_item, item_rc, ch) + } + + fn font_metrics( + &self, + font_request: i_slint_core::graphics::FontRequest, + ) -> i_slint_core::items::FontMetrics { + self.renderer.font_metrics(font_request) + } + + fn text_input_byte_offset_for_position( + &self, + text_input: Pin<&i_slint_core::items::TextInput>, + item_rc: &i_slint_core::items::ItemRc, + pos: i_slint_core::lengths::LogicalPoint, + ) -> usize { + self.renderer.text_input_byte_offset_for_position(text_input, item_rc, pos) + } + + fn text_input_cursor_rect_for_byte_offset( + &self, + text_input: Pin<&i_slint_core::items::TextInput>, + item_rc: &i_slint_core::items::ItemRc, + byte_offset: usize, + ) -> i_slint_core::lengths::LogicalRect { + self.renderer.text_input_cursor_rect_for_byte_offset(text_input, item_rc, byte_offset) + } + + fn register_font_from_memory( + &self, + data: &'static [u8], + ) -> Result<(), Box> { + self.renderer.register_font_from_memory(data) + } + + fn register_font_from_path( + &self, + path: &std::path::Path, + ) -> Result<(), Box> { + self.renderer.register_font_from_path(path) + } + + fn set_rendering_notifier( + &self, + callback: Box, + ) -> Result<(), i_slint_core::api::SetRenderingNotifierError> { + self.renderer.set_rendering_notifier(callback) + } + + fn free_graphics_resources( + &self, + component: i_slint_core::item_tree::ItemTreeRef, + items: &mut dyn Iterator>>, + ) -> Result<(), PlatformError> { + self.renderer.free_graphics_resources(component, items) + } + + fn set_window_adapter(&self, window_adapter: &Rc) { + self.renderer.set_window_adapter(window_adapter) + } + + fn window_adapter(&self) -> Option> { + RendererSealed::window_adapter(&self.renderer) + } + + fn resize(&self, size: i_slint_core::api::PhysicalSize) -> Result<(), PlatformError> { + self.renderer.resize(size) + } + + fn take_snapshot( + &self, + ) -> Result< + i_slint_core::graphics::SharedPixelBuffer, + PlatformError, + > { + self.renderer.take_snapshot() + } + + fn supports_transformations(&self) -> bool { + self.renderer.supports_transformations() + } +} diff --git a/third_party/wgpu-hal-29.0.4/.cargo_vcs_info.json b/third_party/wgpu-hal-29.0.4/.cargo_vcs_info.json new file mode 100644 index 0000000..e241504 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/.cargo_vcs_info.json @@ -0,0 +1,6 @@ +{ + "git": { + "sha1": "e99f5305ded96ff7006f0714d043a7f735bd45c2" + }, + "path_in_vcs": "wgpu-hal" +} \ No newline at end of file diff --git a/third_party/wgpu-hal-29.0.4/Cargo.toml b/third_party/wgpu-hal-29.0.4/Cargo.toml new file mode 100644 index 0000000..f254618 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/Cargo.toml @@ -0,0 +1,520 @@ +# THIS FILE IS AUTOMATICALLY GENERATED BY CARGO +# +# When uploading crates to the registry Cargo will automatically +# "normalize" Cargo.toml files for maximal compatibility +# with all versions of Cargo and also rewrite `path` dependencies +# to registry (e.g., crates.io) dependencies. +# +# If you are reading this file be aware that the original Cargo.toml +# will likely look very different (and much more reasonable). +# See Cargo.toml.orig for the original contents. + +[package] +edition = "2021" +rust-version = "1.87" +name = "wgpu-hal" +version = "29.0.4" +authors = ["gfx-rs developers"] +build = "build.rs" +autolib = false +autobins = false +autoexamples = false +autotests = false +autobenches = false +description = "Hardware abstraction layer for wgpu, the cross-platform, safe, pure-rust graphics API" +homepage = "https://wgpu.rs/" +readme = "README.md" +keywords = ["graphics"] +license = "MIT OR Apache-2.0" +repository = "https://github.com/gfx-rs/wgpu" + +[package.metadata.docs.rs] +features = [ + "metal", + "vulkan", + "gles", + "renderdoc", +] +rustdoc-args = [ + "--cfg", + "docsrs", +] +targets = [ + "x86_64-unknown-linux-gnu", + "x86_64-apple-darwin", + "x86_64-pc-windows-msvc", + "wasm32-unknown-unknown", +] + +[package.metadata.cargo-machete] +ignored = ["cfg_aliases"] + +[features] +device_lost_panic = [] +dx12 = [ + "dep:arrayvec", + "dep:bit-set", + "dep:bytemuck", + "dep:gpu-allocator", + "dep:hashbrown", + "dep:libloading", + "dep:once_cell", + "dep:ordered-float", + "dep:parking_lot", + "dep:profiling", + "dep:range-alloc", + "dep:windows-core", + "gpu-allocator/d3d12", + "naga/hlsl-out", + "once_cell/std", + "windows/Win32_Devices_DeviceAndDriverInstallation", + "windows/Win32_Graphics_Direct3D_Dxc", + "windows/Win32_Graphics_Direct3D_Fxc", + "windows/Win32_Graphics_Direct3D", + "windows/Win32_Graphics_Direct3D12", + "windows/Win32_Graphics_DirectComposition", + "windows/Win32_Graphics_Dxgi_Common", + "windows/Win32_Security", + "windows/Win32_System_Diagnostics_Debug", + "windows/Win32_System_Kernel", + "windows/Win32_System_Performance", + "windows/Win32_System_Threading", + "windows/Win32_UI_WindowsAndMessaging", +] +fragile-send-sync-non-atomic-wasm = ["wgpu-types/fragile-send-sync-non-atomic-wasm"] +gles = [ + "dep:arrayvec", + "dep:bytemuck", + "dep:glow", + "dep:glutin_wgl_sys", + "dep:hashbrown", + "dep:js-sys", + "dep:khronos-egl", + "dep:libloading", + "dep:ndk-sys", + "dep:objc2", + "dep:parking_lot", + "dep:profiling", + "dep:wasm-bindgen", + "dep:wayland-sys", + "dep:web-sys", + "dep:windows-result", + "naga/glsl-out", + "wgpu-types/web", + "windows-result/std", + "windows/Win32_Graphics_Gdi", + "windows/Win32_Graphics_OpenGL", + "windows/Win32_System_LibraryLoader", + "windows/Win32_UI_WindowsAndMessaging", +] +internal_error_panic = [] +metal = [ + "naga/msl-out", + "dep:arrayvec", + "dep:block2", + "dep:bytemuck", + "dep:hashbrown", + "dep:libc", + "dep:objc2", + "dep:objc2-core-foundation", + "dep:objc2-foundation", + "dep:objc2-metal", + "dep:objc2-quartz-core", + "dep:parking_lot", + "dep:profiling", + "dep:smallvec", + "dep:raw-window-metal", +] +portable-atomic = [ + "dep:portable-atomic", + "dep:portable-atomic-util", +] +renderdoc = [ + "dep:libloading", + "dep:renderdoc-sys", +] +static-dxc = ["dep:mach-dxcompiler-rs"] +validation_canary = ["dep:parking_lot"] +vulkan = [ + "naga/spv-out", + "dep:android_system_properties", + "dep:arrayvec", + "dep:ash", + "dep:bytemuck", + "dep:gpu-descriptor", + "dep:hashbrown", + "dep:libc", + "dep:libloading", + "dep:ordered-float", + "dep:parking_lot", + "dep:profiling", + "dep:raw-window-metal", + "dep:smallvec", + "dep:windows", + "gpu-allocator/vulkan", + "windows/Win32", +] + +[lib] +name = "wgpu_hal" +path = "src/lib.rs" + +[[example]] +name = "halmark" +path = "examples/halmark/main.rs" + +[[example]] +name = "raw-gles" +path = "examples/raw-gles.rs" +required-features = ["gles"] + +[[example]] +name = "ray-traced-triangle" +path = "examples/ray-traced-triangle/main.rs" + +[dependencies.arrayvec] +version = "0.7.1" +optional = true +default-features = false + +[dependencies.bitflags] +version = "2.9" + +[dependencies.bytemuck] +version = "1.22" +features = [ + "extern_crate_alloc", + "min_const_generics", + "derive", +] +optional = true + +[dependencies.cfg-if] +version = "1" + +[dependencies.glow] +version = "0.17" +optional = true + +[dependencies.hashbrown] +version = "0.16" +features = [ + "default-hasher", + "inline-more", +] +optional = true +default-features = false + +[dependencies.log] +version = "0.4.29" + +[dependencies.naga] +version = "29.0.1" + +[dependencies.ordered-float] +version = ">=3, <6.0" +optional = true +default-features = false + +[dependencies.parking_lot] +version = "0.12.3" +optional = true + +[dependencies.profiling] +version = "1.0.1" +optional = true +default-features = false + +[dependencies.raw-window-handle] +version = "0.6.2" +default-features = false + +[dependencies.thiserror] +version = "2.0.12" +default-features = false + +[dependencies.wgpu-naga-bridge] +version = "29.0.1" + +[dependencies.wgpu-types] +version = "29.0.1" +default-features = false + +[dev-dependencies.env_logger] +version = "0.11" +default-features = false + +[dev-dependencies.glam] +version = "0.32.0" + +[dev-dependencies.naga] +version = "29.0.1" +features = [ + "wgsl-in", + "termcolor", +] + +[dev-dependencies.winit] +version = "0.30.8" +features = ["android-native-activity"] + +[build-dependencies.cfg_aliases] +version = "0.2.1" + +[target.'cfg(all(target_arch = "wasm32", not(target_os = "emscripten")))'.dependencies.js-sys] +version = "0.3.85" +optional = true +default-features = true + +[target.'cfg(all(target_arch = "wasm32", not(target_os = "emscripten")))'.dependencies.wasm-bindgen] +version = "0.2.108" +optional = true +default-features = false + +[target.'cfg(all(target_arch = "wasm32", not(target_os = "emscripten")))'.dependencies.web-sys] +version = "0.3.85" +features = [ + "default", + "Window", + "HtmlCanvasElement", + "WebGl2RenderingContext", + "OffscreenCanvas", +] +optional = true +default-features = false + +[target.'cfg(all(windows, not(target_arch = "aarch64"), target_env = "msvc"))'.dependencies.mach-dxcompiler-rs] +version = "0.1.4" +optional = true +default-features = false + +[target.'cfg(any(not(target_has_atomic = "64"), not(target_has_atomic = "ptr")))'.dependencies.portable-atomic] +version = "1.10" +optional = true + +[target.'cfg(not(any(target_arch = "wasm32", target_os = "ios", target_os = "visionos", target_env = "ohos")))'.dev-dependencies.glutin] +version = "0.32" +features = [ + "egl", + "wgl", + "wayland", + "x11", +] +default-features = false + +[target.'cfg(not(any(target_arch = "wasm32", target_os = "ios", target_os = "visionos", target_env = "ohos")))'.dev-dependencies.glutin-winit] +version = "0.5" +features = [ + "egl", + "wgl", + "wayland", + "x11", +] +default-features = false + +[target.'cfg(not(target_arch = "wasm32"))'.dependencies.ash] +version = "0.38" +optional = true + +[target.'cfg(not(target_arch = "wasm32"))'.dependencies.gpu-allocator] +version = "0.28" +features = ["hashbrown"] +optional = true +default-features = false + +[target.'cfg(not(target_arch = "wasm32"))'.dependencies.gpu-descriptor] +version = "0.3.2" +optional = true + +[target.'cfg(not(target_arch = "wasm32"))'.dependencies.khronos-egl] +version = "6" +features = ["dynamic"] +optional = true + +[target.'cfg(not(target_arch = "wasm32"))'.dependencies.libloading] +version = "0.8" +optional = true + +[target.'cfg(not(target_arch = "wasm32"))'.dependencies.renderdoc-sys] +version = "1" +optional = true + +[target.'cfg(not(target_arch = "wasm32"))'.dependencies.smallvec] +version = "1.14" +features = ["union"] +optional = true + +[target.'cfg(not(target_has_atomic = "ptr"))'.dependencies.portable-atomic-util] +version = "0.2.4" +features = ["alloc"] +optional = true + +[target.'cfg(target_os = "android")'.dependencies.android_system_properties] +version = "0.1.1" +optional = true + +[target.'cfg(target_os = "android")'.dependencies.ndk-sys] +version = "0.6" +optional = true + +[target.'cfg(target_os = "emscripten")'.dependencies.khronos-egl] +version = "6" +features = [ + "static", + "no-pkg-config", +] +optional = true + +[target.'cfg(target_os = "emscripten")'.dependencies.libloading] +version = "0.8" +optional = true + +[target.'cfg(target_vendor = "apple")'.dependencies.block2] +version = "0.6.2" +optional = true + +[target.'cfg(target_vendor = "apple")'.dependencies.bytemuck] +version = "1.22" +features = [ + "extern_crate_alloc", + "min_const_generics", +] +optional = true + +[target.'cfg(target_vendor = "apple")'.dependencies.objc2] +version = "0.6.3" +optional = true + +[target.'cfg(target_vendor = "apple")'.dependencies.objc2-core-foundation] +version = "0.3.2" +features = [ + "std", + "CFCGTypes", +] +optional = true +default-features = false + +[target.'cfg(target_vendor = "apple")'.dependencies.objc2-foundation] +version = "0.3.2" +features = [ + "std", + "NSError", + "NSProcessInfo", + "NSRange", + "NSString", +] +optional = true +default-features = false + +[target.'cfg(target_vendor = "apple")'.dependencies.objc2-metal] +version = "0.3.2" +features = [ + "std", + "block2", + "MTLAllocation", + "MTLBlitCommandEncoder", + "MTLBlitPass", + "MTLBuffer", + "MTLCaptureManager", + "MTLCaptureScope", + "MTLCommandBuffer", + "MTLCommandEncoder", + "MTLCommandQueue", + "MTLComputeCommandEncoder", + "MTLComputePass", + "MTLComputePipeline", + "MTLCounters", + "MTLDepthStencil", + "MTLDevice", + "MTLDrawable", + "MTLEvent", + "MTLLibrary", + "MTLPipeline", + "MTLPixelFormat", + "MTLRenderCommandEncoder", + "MTLRenderPass", + "MTLRenderPipeline", + "MTLResource", + "MTLSampler", + "MTLStageInputOutputDescriptor", + "MTLTexture", + "MTLTypes", + "MTLVertexDescriptor", + "MTLAccelerationStructure", + "MTLAccelerationStructureTypes", + "MTLAccelerationStructureCommandEncoder", + "MTLResidencySet", +] +optional = true +default-features = false + +[target.'cfg(target_vendor = "apple")'.dependencies.objc2-quartz-core] +version = "0.3.2" +features = [ + "std", + "objc2-core-foundation", + "CALayer", + "CAMetalLayer", + "objc2-metal", +] +optional = true +default-features = false + +[target.'cfg(target_vendor = "apple")'.dependencies.raw-window-metal] +version = "1.0" +optional = true + +[target."cfg(unix)".dependencies.libc] +version = "0.2.172" +optional = true +default-features = false + +[target."cfg(unix)".dependencies.wayland-sys] +version = "0.31.3" +features = [ + "client", + "dlopen", + "egl", +] +optional = true + +[target."cfg(windows)".dependencies.bit-set] +version = "0.9" +optional = true +default-features = false + +[target."cfg(windows)".dependencies.glutin_wgl_sys] +version = "0.6" +optional = true + +[target."cfg(windows)".dependencies.once_cell] +version = "1.21" +optional = true +default-features = false + +[target."cfg(windows)".dependencies.range-alloc] +version = "0.1" +optional = true + +[target."cfg(windows)".dependencies.windows] +version = "0.62" +optional = true +default-features = false + +[target."cfg(windows)".dependencies.windows-core] +version = "0.62" +optional = true +default-features = false + +[target."cfg(windows)".dependencies.windows-result] +version = "0.4" +optional = true +default-features = false + +[lints.rust.unexpected_cfgs] +level = "warn" +priority = 0 +check-cfg = [ + 'cfg(feature, values("cargo-clippy"))', + "cfg(web_sys_unstable_apis)", +] diff --git a/third_party/wgpu-hal-29.0.4/Cargo.toml.orig b/third_party/wgpu-hal-29.0.4/Cargo.toml.orig new file mode 100644 index 0000000..3026526 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/Cargo.toml.orig @@ -0,0 +1,355 @@ +[package] +name = "wgpu-hal" +version.workspace = true +authors.workspace = true +edition.workspace = true +description = "Hardware abstraction layer for wgpu, the cross-platform, safe, pure-rust graphics API" +homepage.workspace = true +repository.workspace = true +keywords.workspace = true +license.workspace = true + +# Override the workspace's `rust-version` key. `wgpu-core` and its dependencies +# have a less strict MSRV, to allow firefox more leeway in updating their Rust toolchain. +# +# See the repo README for more information on MSRV policy. +rust-version = "1.87" + +[package.metadata.docs.rs] +# Ideally we would enable all the features. +# +# However, the dx12 features fail to be documented because the docs.rs runner cross-compiling under +# x86_64-unknown-linux-gnu cannot compile in that environment at the moment. +features = ["metal", "vulkan", "gles", "renderdoc"] +rustdoc-args = ["--cfg", "docsrs"] +targets = [ + "x86_64-unknown-linux-gnu", + "x86_64-apple-darwin", + "x86_64-pc-windows-msvc", + "wasm32-unknown-unknown", +] + +[package.metadata.cargo-machete] +# Cargo machete can't check build.rs dependencies. See https://github.com/bnjbvr/cargo-machete/issues/100 +ignored = ["cfg_aliases"] + +[lints.rust] +unexpected_cfgs = { level = "warn", check-cfg = [ + 'cfg(feature, values("cargo-clippy"))', # objc's `msg_send` macro injects this in our code https://github.com/SSheldon/rust-objc/issues/125 + 'cfg(web_sys_unstable_apis)', # web-sys uses this +] } + +[lib] + +[features] + +######################## +### Backend Features ### +######################## + +# The interaction of features between wgpu-core and wgpu-hal is a bit nuanced to get +# the desired behavior on all platforms. +# +# At the wgpu-hal level the features are defined to enable the backends on all platforms +# that can compile the backend. Vulkan for example will have an effect on Windows, Mac, Linux, and Android. +# This is done with target conditional dependencies in wgpu-hal. This allows `--all-features` +# to compile on all platforms. +# +# wgpu-core's features are defined to enable the backends on their "default" platforms. For example we +# exclude the Vulkan backend on MacOS unless a separate feature `vulkan-portability` is enabled. In response +# to these features, it enables features of platform specific crates. For example, the `vulkan` feature in wgpu-core +# enables the `vulkan` feature in `wgpu-core-deps-windows-linux-android` which in turn enables the +# `vulkan` feature in `wgpu-hal` _only_ on those platforms. If you enable the `vulkan-portability` feature, it +# will enable the `vulkan` feature in `wgpu-core-deps-apple`. The only way to do this is unfortunately to have +# a separate crate for each platform category that participates in the feature unification. +# +# This trick doesn't work at the `wgpu` level, because the `wgpu` -> `wgpu-core` dependency is conditional, +# making the Cargo.toml signifigantly more complicated in all areas. +# +# See https://github.com/gfx-rs/wgpu/issues/3514, https://github.com/gfx-rs/wgpu/pull/7076, +# and https://github.com/rust-lang/cargo/issues/1197 for more information. + +## Enables the Metal backend when targeting Apple platforms. +metal = [ + # Metal is only available on Apple platforms, therefore request MSL output also only if we target an Apple platform. + "naga/msl-out", + "dep:arrayvec", + "dep:block2", + "dep:bytemuck", + "dep:hashbrown", + "dep:libc", + "dep:objc2", + "dep:objc2-core-foundation", + "dep:objc2-foundation", + "dep:objc2-metal", + "dep:objc2-quartz-core", + "dep:parking_lot", + "dep:profiling", + "dep:smallvec", + "dep:raw-window-metal", +] +vulkan = [ + "naga/spv-out", + "dep:android_system_properties", + "dep:arrayvec", + "dep:ash", + "dep:bytemuck", + "dep:gpu-descriptor", + "dep:hashbrown", + "dep:libc", + "dep:libloading", + "dep:ordered-float", + "dep:parking_lot", + "dep:profiling", + "dep:raw-window-metal", + "dep:smallvec", + "dep:windows", + "gpu-allocator/vulkan", + "windows/Win32", +] +gles = [ + "dep:arrayvec", + "dep:bytemuck", + "dep:glow", + "dep:glutin_wgl_sys", + "dep:hashbrown", + "dep:js-sys", + "dep:khronos-egl", + "dep:libloading", + "dep:ndk-sys", + "dep:objc2", + "dep:parking_lot", + "dep:profiling", + "dep:wasm-bindgen", + "dep:wayland-sys", + "dep:web-sys", + "dep:windows-result", + "naga/glsl-out", + "wgpu-types/web", + "windows-result/std", # Need to enable this for the `Error` impl on `windows_result::Error` to work. + "windows/Win32_Graphics_Gdi", + "windows/Win32_Graphics_OpenGL", + "windows/Win32_System_LibraryLoader", + "windows/Win32_UI_WindowsAndMessaging", +] +## Enables the DX12 backend when targeting Windows. +dx12 = [ + "dep:arrayvec", + "dep:bit-set", + "dep:bytemuck", + "dep:gpu-allocator", + "dep:hashbrown", + "dep:libloading", + "dep:once_cell", + "dep:ordered-float", + "dep:parking_lot", + "dep:profiling", + "dep:range-alloc", + "dep:windows-core", + "gpu-allocator/d3d12", + "naga/hlsl-out", + "once_cell/std", + "windows/Win32_Devices_DeviceAndDriverInstallation", + "windows/Win32_Graphics_Direct3D_Dxc", + "windows/Win32_Graphics_Direct3D_Fxc", + "windows/Win32_Graphics_Direct3D", + "windows/Win32_Graphics_Direct3D12", + "windows/Win32_Graphics_DirectComposition", + "windows/Win32_Graphics_Dxgi_Common", + "windows/Win32_Security", + "windows/Win32_System_Diagnostics_Debug", + "windows/Win32_System_Kernel", + "windows/Win32_System_Performance", + "windows/Win32_System_Threading", + "windows/Win32_UI_WindowsAndMessaging", +] + +########################### +### Misc Other Features ### +########################### + +static-dxc = ["dep:mach-dxcompiler-rs"] +renderdoc = ["dep:libloading", "dep:renderdoc-sys"] +fragile-send-sync-non-atomic-wasm = [ + "wgpu-types/fragile-send-sync-non-atomic-wasm", +] +portable-atomic = ["dep:portable-atomic", "dep:portable-atomic-util"] + +################################### +### Internal Debugging Features ### +################################### + +# Panic when running into a device lost error (for debugging purposes). +# Only affects the d3d12 and vulkan backends. +device_lost_panic = [] +# Panic when running into an internal error other than out-of-memory and device lost +# (for debugging purposes). +# +# Only affects the d3d12 and vulkan backends. +internal_error_panic = [] +# Tracks validation errors in a `VALIDATION_CANARY` static. +validation_canary = ["dep:parking_lot"] + +[[example]] +name = "halmark" + +[[example]] +name = "raw-gles" +required-features = ["gles"] + +##################### +### Platform: All ### +##################### + +[dependencies] +naga.workspace = true +wgpu-naga-bridge.workspace = true +wgpu-types = { workspace = true, default-features = false } + +# Dependencies in the lib and empty backend +bitflags.workspace = true +cfg-if.workspace = true +log = { workspace = true } +parking_lot = { workspace = true, optional = true } +raw-window-handle.workspace = true +thiserror.workspace = true + +# Target agnostic dependencies used only in backends. +arrayvec = { workspace = true, optional = true } +bytemuck = { workspace = true, optional = true, features = ["derive"] } +hashbrown = { workspace = true, optional = true } +ordered-float = { workspace = true, optional = true } +profiling = { workspace = true, optional = true, default-features = false } + +# Backend: GLES +glow = { workspace = true, optional = true } + +######################## +### Platform: Native ### +######################## + +[target.'cfg(not(target_arch = "wasm32"))'.dependencies] +# Backend: Vulkan and Dx12 +gpu-allocator = { workspace = true, optional = true } +# Backend: Vulkan +ash = { workspace = true, optional = true } +gpu-descriptor = { workspace = true, optional = true } +smallvec = { workspace = true, optional = true, features = ["union"] } +# Backend: GLES +khronos-egl = { workspace = true, features = ["dynamic"], optional = true } +libloading = { workspace = true, optional = true } +renderdoc-sys = { workspace = true, optional = true } + +########################## +### Platform: All Unix ### +########################## + +[target.'cfg(unix)'.dependencies] +# Backend: Vulkan +libc = { workspace = true, optional = true } +# backend: GLES +wayland-sys = { version = "0.31.3", features = [ + "client", + "dlopen", + "egl", +], optional = true } + +######################### +### Platform: Windows ### +######################### + +[target.'cfg(windows)'.dependencies] +# Backend: Dx12 and GLES +windows = { workspace = true, optional = true } +windows-core = { workspace = true, optional = true } +windows-result = { workspace = true, optional = true } +# Backend: Dx12 +bit-set = { workspace = true, optional = true } +range-alloc = { workspace = true, optional = true } +once_cell = { workspace = true, optional = true } +# backend: GLES +glutin_wgl_sys = { workspace = true, optional = true } + +### Platform: x86/x86_64 Windows ### +# This doesn't support aarch64. See https://github.com/gfx-rs/wgpu/issues/6860. +# This doesn't support x86_64-pc-windows-gnu. See https://github.com/gfx-rs/wgpu/issues/8303 +# +# ⚠️ Keep in sync with static_dxc cfg in build.rs and cfg_alias in `wgpu` crate ⚠️ +[target.'cfg(all(windows, not(target_arch = "aarch64"), target_env = "msvc"))'.dependencies] +mach-dxcompiler-rs = { workspace = true, optional = true } + +####################### +### Platform: Apple ### +####################### + +[target.'cfg(target_vendor = "apple")'.dependencies] +# Backend: Metal +block2 = { workspace = true, optional = true } +bytemuck = { workspace = true, optional = true } +objc2 = { workspace = true, optional = true } +objc2-core-foundation = { workspace = true, optional = true } +objc2-foundation = { workspace = true, optional = true } +objc2-metal = { workspace = true, optional = true } +objc2-quartz-core = { workspace = true, optional = true } + +# backend: Metal + Vulkan +raw-window-metal = { workspace = true, optional = true } + +######################### +### Platform: Android ### +######################### + +[target.'cfg(target_os = "android")'.dependencies] +android_system_properties = { workspace = true, optional = true } +ndk-sys = { workspace = true, optional = true } + +############################# +### Platform: Webassembly ### +############################# + +[target.'cfg(all(target_arch = "wasm32", not(target_os = "emscripten")))'.dependencies] +# Backend: GLES +wasm-bindgen = { workspace = true, optional = true } +web-sys = { workspace = true, optional = true, features = [ + "default", + "Window", + "HtmlCanvasElement", + "WebGl2RenderingContext", + "OffscreenCanvas", +] } +js-sys = { workspace = true, optional = true, default-features = true } + +############################ +### Platform: Emscripten ### +############################ + +[target.'cfg(target_os = "emscripten")'.dependencies] +# Backend: GLES +khronos-egl = { workspace = true, optional = true, features = [ + "static", + "no-pkg-config", +] } +# Note: it's unused by emscripten, but we keep it to have single code base in egl.rs +libloading = { workspace = true, optional = true } + +[target.'cfg(any(not(target_has_atomic = "64"), not(target_has_atomic = "ptr")))'.dependencies] +portable-atomic = { workspace = true, optional = true } + +[target.'cfg(not(target_has_atomic = "ptr"))'.dependencies] +portable-atomic-util = { workspace = true, features = [ + "alloc", +], optional = true } + +[build-dependencies] +cfg_aliases.workspace = true + +[dev-dependencies] +env_logger.workspace = true +glam.workspace = true # for ray-traced-triangle example +naga = { workspace = true, features = ["wgsl-in", "termcolor"] } +winit.workspace = true # for "halmark" example + +### Platform: Windows + MacOS + Linux for "raw-gles" example ### +[target.'cfg(not(any(target_arch = "wasm32", target_os = "ios", target_os = "visionos", target_env = "ohos")))'.dev-dependencies] +glutin-winit = { workspace = true, features = ["egl", "wgl", "wayland", "x11"] } +glutin = { workspace = true, features = ["egl", "wgl", "wayland", "x11"] } diff --git a/third_party/wgpu-hal-29.0.4/LICENSE.APACHE b/third_party/wgpu-hal-29.0.4/LICENSE.APACHE new file mode 100644 index 0000000..d9a10c0 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/LICENSE.APACHE @@ -0,0 +1,176 @@ + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS diff --git a/third_party/wgpu-hal-29.0.4/LICENSE.MIT b/third_party/wgpu-hal-29.0.4/LICENSE.MIT new file mode 100644 index 0000000..8d02e4d --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/LICENSE.MIT @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2025 The gfx-rs developers + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/third_party/wgpu-hal-29.0.4/README.md b/third_party/wgpu-hal-29.0.4/README.md new file mode 100644 index 0000000..c1abf60 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/README.md @@ -0,0 +1,120 @@ +# `wgpu_hal`: a cross-platform unsafe graphics abstraction + +This crate defines a set of traits abstracting over modern graphics APIs, +with implementations ("backends") for Vulkan, Metal, Direct3D, and GL. + +`wgpu_hal` is a spiritual successor to +[gfx-hal](https://github.com/gfx-rs/gfx), but with reduced scope, and +oriented towards WebGPU implementation goals. It has no overhead for +validation or tracking, and the API translation overhead is kept to the bare +minimum by the design of WebGPU. This API can be used for resource-demanding +applications and engines. + +The `wgpu_hal` crate's main design choices: + +- Our traits are meant to be *portable*: proper use + should get equivalent results regardless of the backend. + +- Our traits' contracts are *unsafe*: implementations perform minimal + validation, if any, and incorrect use will often cause undefined behavior. + This allows us to minimize the overhead we impose over the underlying + graphics system. If you need safety, the [`wgpu-core`] crate provides a + safe API for driving `wgpu_hal`, implementing all necessary validation, + resource state tracking, and so on. (Note that `wgpu-core` is designed for + use via FFI; the [`wgpu`] crate provides more idiomatic Rust bindings for + `wgpu-core`.) Or, you can do your own validation. + +- In the same vein, returned errors *only cover cases the user can't + anticipate*, like running out of memory or losing the device. Any errors + that the user could reasonably anticipate are their responsibility to + avoid. For example, `wgpu_hal` returns no error for mapping a buffer that's + not mappable: as the buffer creator, the user should already know if they + can map it. + +- We use *static dispatch*. The traits are not + generally object-safe. You must select a specific backend type + like [`vulkan::Api`] or [`metal::Api`], and then use that + according to the main traits, or call backend-specific methods. + +- We use *idiomatic Rust parameter passing*, + taking objects by reference, returning them by value, and so on, + unlike `wgpu-core`, which refers to objects by ID. + +- We map buffer contents *persistently*. This means that the buffer + can remain mapped on the CPU while the GPU reads or writes to it. + You must explicitly indicate when data might need to be + transferred between CPU and GPU, if `wgpu_hal` indicates that the + mapping is not coherent (that is, automatically synchronized + between the two devices). + +- You must record *explicit barriers* between different usages of a + resource. For example, if a buffer is written to by a compute + shader, and then used as and index buffer to a draw call, you + must use [`CommandEncoder::transition_buffers`] between those two + operations. + +- Pipeline layouts are *explicitly specified* when setting bind + group. Incompatible layouts disturb groups bound at higher indices. + +- The API *accepts collections as iterators*, to avoid forcing the user to + store data in particular containers. The implementation doesn't guarantee + that any of the iterators are drained, unless stated otherwise by the + function documentation. For this reason, we recommend that iterators don't + do any mutating work. + +Unfortunately, `wgpu_hal`'s safety requirements are not fully documented. +Ideally, all trait methods would have doc comments setting out the +requirements users must meet to ensure correct and portable behavior. If you +are aware of a specific requirement that a backend imposes that is not +ensured by the traits' documented rules, please file an issue. Or, if you are +a capable technical writer, please file a pull request! + +[`wgpu-core`]: https://crates.io/crates/wgpu-core +[`wgpu`]: https://crates.io/crates/wgpu +[`vulkan::Api`]: vulkan/struct.Api.html +[`metal::Api`]: metal/struct.Api.html + +## Primary backends + +The `wgpu_hal` crate has full-featured backends implemented on the following +platform graphics APIs: + +- Vulkan, available on Linux, Android, and Windows, using the [`ash`] crate's + Vulkan bindings. It's also available on macOS, if you install [MoltenVK]. + +- Metal on macOS, using the [`metal`] crate's bindings. + +- Direct3D 12 on Windows, using the [`windows`] crate's bindings. + +[`ash`]: https://crates.io/crates/ash +[MoltenVK]: https://github.com/KhronosGroup/MoltenVK +[`metal`]: https://crates.io/crates/metal +[`windows`]: https://crates.io/crates/windows + +## Secondary backends + +The `wgpu_hal` crate has a partial implementation based on the following +platform graphics API: + +- The GL backend is available anywhere OpenGL, OpenGL ES, or WebGL are + available. See the [`gles`] module documentation for details. + +[`gles`]: gles/index.html + +You can see what capabilities an adapter is missing by checking the +[`DownlevelCapabilities`][tdc] in [`ExposedAdapter::capabilities`], available +from [`Instance::enumerate_adapters`]. + +The API is generally designed to fit the primary backends better than the +secondary backends, so the latter may impose more overhead. + +[tdc]: wgt::DownlevelCapabilities + +## Debugging + +Most of the information on the wiki [Debugging wgpu Applications][wiki-debug] +page still applies to this API, with the exception of API tracing/replay +functionality, which is only available in `wgpu-core`. + +[wiki-debug]: https://github.com/gfx-rs/wgpu/wiki/Debugging-wgpu-Applications + diff --git a/third_party/wgpu-hal-29.0.4/build.rs b/third_party/wgpu-hal-29.0.4/build.rs new file mode 100644 index 0000000..7bb56e9 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/build.rs @@ -0,0 +1,32 @@ +fn main() { + cfg_aliases::cfg_aliases! { + native: { not(target_arch = "wasm32") }, + send_sync: { any( + not(target_arch = "wasm32"), + all(feature = "fragile-send-sync-non-atomic-wasm", not(target_feature = "atomics")) + ) }, + webgl: { all(target_arch = "wasm32", not(target_os = "emscripten"), gles) }, + Emscripten: { all(target_os = "emscripten", gles) }, + dx12: { all(target_os = "windows", feature = "dx12") }, + gles: { all(feature = "gles") }, + // Within the GL ES backend, use `std` and be Send + Sync only if we are using a target + // that, among the ones where the GL ES backend is supported, has `std`. + gles_with_std: { all( + feature = "gles", + any( + not(target_arch = "wasm32"), + // Accept wasm32-unknown-unknown, which uniquely has a stub `std` + all(target_vendor = "unknown", target_os = "unknown"), + // Accept wasm32-unknown-emscripten and similar, which has a real `std` + target_os = "emscripten" + ) + ) }, + metal: { all(target_vendor = "apple", feature = "metal") }, + vulkan: { all(not(target_arch = "wasm32"), feature = "vulkan") }, + any_backend: { any(dx12, metal, vulkan, gles) }, + // ⚠️ Keep in sync with target.cfg() definition in Cargo.toml and cfg_alias in `wgpu` crate ⚠️ + static_dxc: { all(target_os = "windows", feature = "static-dxc", not(target_arch = "aarch64"), target_env = "msvc") }, + supports_64bit_atomics: { target_has_atomic = "64" }, + supports_ptr_atomics: { target_has_atomic = "ptr" } + } +} diff --git a/third_party/wgpu-hal-29.0.4/examples/halmark/main.rs b/third_party/wgpu-hal-29.0.4/examples/halmark/main.rs new file mode 100644 index 0000000..9983de9 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/examples/halmark/main.rs @@ -0,0 +1,918 @@ +//! This example shows basic usage of wgpu-hal by rendering +//! a ton of moving sprites, each with a separate texture and draw call. +extern crate wgpu_hal as hal; + +use hal::{ + Adapter as _, CommandEncoder as _, Device as _, Instance as _, Queue as _, Surface as _, +}; +use raw_window_handle::{HasDisplayHandle, HasWindowHandle}; +use winit::{ + application::ApplicationHandler, + event::{ElementState, KeyEvent, WindowEvent}, + event_loop::{ActiveEventLoop, ControlFlow}, + keyboard::{Key, NamedKey}, + window::Window, +}; + +use std::{ + borrow::{Borrow, Cow}, + iter, + num::NonZeroU64, + ptr, + time::Instant, +}; + +const MAX_BUNNIES: usize = 1 << 20; +const BUNNY_SIZE: f32 = 0.15 * 256.0; +const GRAVITY: f32 = -9.8 * 100.0; +const MAX_VELOCITY: f32 = 750.0; +const DESIRED_MAX_LATENCY: u32 = 2; + +#[repr(C)] +#[derive(Clone, Copy)] +struct Globals { + mvp: [[f32; 4]; 4], + size: [f32; 2], + pad: [f32; 2], +} + +#[repr(C, align(256))] +#[derive(Clone, Copy)] +struct Locals { + position: [f32; 2], + velocity: [f32; 2], + color: u32, + _pad: u32, +} + +struct ExecutionContext { + encoder: A::CommandEncoder, + fence: A::Fence, + fence_value: hal::FenceValue, + used_views: Vec, + used_cmd_bufs: Vec, + frames_recorded: usize, +} + +impl ExecutionContext { + unsafe fn wait_and_clear(&mut self, device: &A::Device) { + device.wait(&self.fence, self.fence_value, None).unwrap(); + self.encoder.reset_all(self.used_cmd_bufs.drain(..)); + for view in self.used_views.drain(..) { + device.destroy_texture_view(view); + } + self.frames_recorded = 0; + } +} + +#[allow(dead_code)] +struct Example { + instance: A::Instance, + adapter: A::Adapter, + surface: A::Surface, + surface_format: wgpu_types::TextureFormat, + device: A::Device, + queue: A::Queue, + global_group: A::BindGroup, + local_group: A::BindGroup, + global_group_layout: A::BindGroupLayout, + local_group_layout: A::BindGroupLayout, + pipeline_layout: A::PipelineLayout, + shader: A::ShaderModule, + pipeline: A::RenderPipeline, + bunnies: Vec, + local_buffer: A::Buffer, + local_alignment: u32, + global_buffer: A::Buffer, + sampler: A::Sampler, + texture: A::Texture, + texture_view: A::TextureView, + contexts: Vec>, + context_index: usize, + extent: [u32; 2], + start: Instant, +} + +impl Example { + fn init(window: &winit::window::Window) -> Result> { + // The Instance can be initialized with the DisplayHandle from the EventLoop as well + let raw_display_handle = window.display_handle()?; + + let instance_desc = hal::InstanceDescriptor { + name: "example", + flags: wgpu_types::InstanceFlags::from_build_config().with_env(), + memory_budget_thresholds: wgpu_types::MemoryBudgetThresholds::default(), + // Can't rely on having DXC available, so use FXC instead + backend_options: wgpu_types::BackendOptions::default(), + telemetry: None, + display: Some(raw_display_handle), + }; + let instance = unsafe { A::Instance::init(&instance_desc)? }; + let surface = { + let raw_window_handle = window.window_handle()?.as_raw(); + + unsafe { + instance + .create_surface(raw_display_handle.as_raw(), raw_window_handle) + .unwrap() + } + }; + + let (adapter, capabilities) = unsafe { + let mut adapters = instance.enumerate_adapters(Some(&surface)); + if adapters.is_empty() { + return Err("no adapters found".into()); + } + let exposed = adapters.swap_remove(0); + (exposed.adapter, exposed.capabilities) + }; + + let surface_caps = unsafe { adapter.surface_capabilities(&surface) } + .ok_or("failed to get surface capabilities")?; + log::info!("Surface caps: {surface_caps:#?}"); + + let hal::OpenDevice { device, queue } = unsafe { + adapter + .open( + wgpu_types::Features::empty(), + &wgpu_types::Limits::default(), + &wgpu_types::MemoryHints::default(), + ) + .unwrap() + }; + + let window_size: (u32, u32) = window.inner_size().into(); + let surface_config = hal::SurfaceConfiguration { + maximum_frame_latency: DESIRED_MAX_LATENCY.clamp( + *surface_caps.maximum_frame_latency.start(), + *surface_caps.maximum_frame_latency.end(), + ), + present_mode: wgpu_types::PresentMode::Fifo, + composite_alpha_mode: wgpu_types::CompositeAlphaMode::Opaque, + format: wgpu_types::TextureFormat::Bgra8UnormSrgb, + extent: wgpu_types::Extent3d { + width: window_size.0, + height: window_size.1, + depth_or_array_layers: 1, + }, + usage: wgpu_types::TextureUses::COLOR_TARGET, + view_formats: vec![], + }; + unsafe { + surface.configure(&device, &surface_config).unwrap(); + }; + + let naga_shader = { + let shader_file = std::path::PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("examples") + .join("halmark") + .join("shader.wgsl"); + let source = std::fs::read_to_string(shader_file).unwrap(); + let module = naga::front::wgsl::Frontend::new().parse(&source).unwrap(); + let info = naga::valid::Validator::new( + naga::valid::ValidationFlags::all(), + naga::valid::Capabilities::empty(), + ) + .validate(&module) + .unwrap(); + hal::NagaShader { + module: Cow::Owned(module), + info, + debug_source: None, + } + }; + let shader_desc = hal::ShaderModuleDescriptor { + label: None, + runtime_checks: wgpu_types::ShaderRuntimeChecks::checked(), + }; + let shader = unsafe { + device + .create_shader_module(&shader_desc, hal::ShaderInput::Naga(naga_shader)) + .unwrap() + }; + + let global_bgl_desc = hal::BindGroupLayoutDescriptor { + label: None, + flags: hal::BindGroupLayoutFlags::empty(), + entries: &[ + wgpu_types::BindGroupLayoutEntry { + binding: 0, + visibility: wgpu_types::ShaderStages::VERTEX, + ty: wgpu_types::BindingType::Buffer { + ty: wgpu_types::BufferBindingType::Uniform, + has_dynamic_offset: false, + min_binding_size: wgpu_types::BufferSize::new(size_of::() as _), + }, + count: None, + }, + wgpu_types::BindGroupLayoutEntry { + binding: 1, + visibility: wgpu_types::ShaderStages::FRAGMENT, + ty: wgpu_types::BindingType::Texture { + sample_type: wgpu_types::TextureSampleType::Float { filterable: true }, + view_dimension: wgpu_types::TextureViewDimension::D2, + multisampled: false, + }, + count: None, + }, + wgpu_types::BindGroupLayoutEntry { + binding: 2, + visibility: wgpu_types::ShaderStages::FRAGMENT, + ty: wgpu_types::BindingType::Sampler(wgpu_types::SamplerBindingType::Filtering), + count: None, + }, + ], + }; + + let global_group_layout = + unsafe { device.create_bind_group_layout(&global_bgl_desc).unwrap() }; + + let local_bgl_desc = hal::BindGroupLayoutDescriptor { + label: None, + flags: hal::BindGroupLayoutFlags::empty(), + entries: &[wgpu_types::BindGroupLayoutEntry { + binding: 0, + visibility: wgpu_types::ShaderStages::VERTEX, + ty: wgpu_types::BindingType::Buffer { + ty: wgpu_types::BufferBindingType::Uniform, + has_dynamic_offset: true, + min_binding_size: wgpu_types::BufferSize::new(size_of::() as _), + }, + count: None, + }], + }; + let local_group_layout = + unsafe { device.create_bind_group_layout(&local_bgl_desc).unwrap() }; + + let pipeline_layout_desc = hal::PipelineLayoutDescriptor { + label: None, + flags: hal::PipelineLayoutFlags::empty(), + bind_group_layouts: &[Some(&global_group_layout), Some(&local_group_layout)], + immediate_size: 0, + }; + let pipeline_layout = unsafe { + device + .create_pipeline_layout(&pipeline_layout_desc) + .unwrap() + }; + + let constants = naga::back::PipelineConstants::default(); + let pipeline_desc = hal::RenderPipelineDescriptor { + label: None, + layout: &pipeline_layout, + vertex_processor: hal::VertexProcessor::Standard { + vertex_stage: hal::ProgrammableStage { + module: &shader, + entry_point: "vs_main", + constants: &constants, + zero_initialize_workgroup_memory: true, + }, + vertex_buffers: &[], + }, + fragment_stage: Some(hal::ProgrammableStage { + module: &shader, + entry_point: "fs_main", + constants: &constants, + zero_initialize_workgroup_memory: true, + }), + primitive: wgpu_types::PrimitiveState { + topology: wgpu_types::PrimitiveTopology::TriangleStrip, + ..wgpu_types::PrimitiveState::default() + }, + depth_stencil: None, + multisample: wgpu_types::MultisampleState::default(), + color_targets: &[Some(wgpu_types::ColorTargetState { + format: surface_config.format, + blend: Some(wgpu_types::BlendState::ALPHA_BLENDING), + write_mask: wgpu_types::ColorWrites::default(), + })], + multiview_mask: None, + cache: None, + }; + let pipeline = unsafe { device.create_render_pipeline(&pipeline_desc).unwrap() }; + + let texture_data = [0xFFu8; 4]; + + let staging_buffer_desc = hal::BufferDescriptor { + label: Some("stage"), + size: texture_data.len() as wgpu_types::BufferAddress, + usage: wgpu_types::BufferUses::MAP_WRITE | wgpu_types::BufferUses::COPY_SRC, + memory_flags: hal::MemoryFlags::TRANSIENT | hal::MemoryFlags::PREFER_COHERENT, + }; + let staging_buffer = unsafe { device.create_buffer(&staging_buffer_desc).unwrap() }; + unsafe { + let mapping = device + .map_buffer(&staging_buffer, 0..staging_buffer_desc.size) + .unwrap(); + ptr::copy_nonoverlapping( + texture_data.as_ptr(), + mapping.ptr.as_ptr(), + texture_data.len(), + ); + device.unmap_buffer(&staging_buffer); + assert!(mapping.is_coherent); + } + + let texture_desc = hal::TextureDescriptor { + label: None, + size: wgpu_types::Extent3d { + width: 1, + height: 1, + depth_or_array_layers: 1, + }, + mip_level_count: 1, + sample_count: 1, + dimension: wgpu_types::TextureDimension::D2, + format: wgpu_types::TextureFormat::Rgba8UnormSrgb, + usage: wgpu_types::TextureUses::COPY_DST | wgpu_types::TextureUses::RESOURCE, + memory_flags: hal::MemoryFlags::empty(), + view_formats: vec![], + }; + let texture = unsafe { device.create_texture(&texture_desc).unwrap() }; + + let cmd_encoder_desc = hal::CommandEncoderDescriptor { + label: None, + queue: &queue, + }; + let mut cmd_encoder = unsafe { device.create_command_encoder(&cmd_encoder_desc).unwrap() }; + unsafe { cmd_encoder.begin_encoding(Some("init")).unwrap() }; + { + let buffer_barrier = hal::BufferBarrier { + buffer: &staging_buffer, + usage: hal::StateTransition { + from: wgpu_types::BufferUses::empty(), + to: wgpu_types::BufferUses::COPY_SRC, + }, + }; + let texture_barrier1 = hal::TextureBarrier { + texture: &texture, + range: wgpu_types::ImageSubresourceRange::default(), + usage: hal::StateTransition { + from: wgpu_types::TextureUses::UNINITIALIZED, + to: wgpu_types::TextureUses::COPY_DST, + }, + }; + let texture_barrier2 = hal::TextureBarrier { + texture: &texture, + range: wgpu_types::ImageSubresourceRange::default(), + usage: hal::StateTransition { + from: wgpu_types::TextureUses::COPY_DST, + to: wgpu_types::TextureUses::RESOURCE, + }, + }; + let copy = hal::BufferTextureCopy { + buffer_layout: wgpu_types::TexelCopyBufferLayout { + offset: 0, + bytes_per_row: Some(4), + rows_per_image: None, + }, + texture_base: hal::TextureCopyBase { + origin: wgpu_types::Origin3d::ZERO, + mip_level: 0, + array_layer: 0, + aspect: hal::FormatAspects::COLOR, + }, + size: hal::CopyExtent { + width: 1, + height: 1, + depth: 1, + }, + }; + unsafe { + cmd_encoder.transition_buffers(iter::once(buffer_barrier)); + cmd_encoder.transition_textures(iter::once(texture_barrier1)); + cmd_encoder.copy_buffer_to_texture(&staging_buffer, &texture, iter::once(copy)); + cmd_encoder.transition_textures(iter::once(texture_barrier2)); + } + } + + let sampler_desc = hal::SamplerDescriptor { + label: None, + address_modes: [wgpu_types::AddressMode::ClampToEdge; 3], + mag_filter: wgpu_types::FilterMode::Linear, + min_filter: wgpu_types::FilterMode::Nearest, + mipmap_filter: wgpu_types::MipmapFilterMode::Nearest, + lod_clamp: 0.0..32.0, + compare: None, + anisotropy_clamp: 1, + border_color: None, + }; + let sampler = unsafe { device.create_sampler(&sampler_desc).unwrap() }; + + let globals = Globals { + // cgmath::ortho() projection + mvp: [ + [2.0 / window_size.0 as f32, 0.0, 0.0, 0.0], + [0.0, 2.0 / window_size.1 as f32, 0.0, 0.0], + [0.0, 0.0, 1.0, 0.0], + [-1.0, -1.0, 0.0, 1.0], + ], + size: [BUNNY_SIZE; 2], + pad: [0.0; 2], + }; + + let global_buffer_desc = hal::BufferDescriptor { + label: Some("global"), + size: size_of::() as wgpu_types::BufferAddress, + usage: wgpu_types::BufferUses::MAP_WRITE | wgpu_types::BufferUses::UNIFORM, + memory_flags: hal::MemoryFlags::PREFER_COHERENT, + }; + let global_buffer = unsafe { + let buffer = device.create_buffer(&global_buffer_desc).unwrap(); + let mapping = device + .map_buffer(&buffer, 0..global_buffer_desc.size) + .unwrap(); + ptr::copy_nonoverlapping( + &globals as *const Globals as *const u8, + mapping.ptr.as_ptr(), + size_of::(), + ); + device.unmap_buffer(&buffer); + assert!(mapping.is_coherent); + buffer + }; + + let local_alignment = wgpu_types::math::align_to( + size_of::() as u32, + capabilities.limits.min_uniform_buffer_offset_alignment, + ); + let local_buffer_desc = hal::BufferDescriptor { + label: Some("local"), + size: (MAX_BUNNIES as wgpu_types::BufferAddress) + * (local_alignment as wgpu_types::BufferAddress), + usage: wgpu_types::BufferUses::MAP_WRITE | wgpu_types::BufferUses::UNIFORM, + memory_flags: hal::MemoryFlags::PREFER_COHERENT, + }; + let local_buffer = unsafe { device.create_buffer(&local_buffer_desc).unwrap() }; + + let view_desc = hal::TextureViewDescriptor { + label: None, + format: texture_desc.format, + dimension: wgpu_types::TextureViewDimension::D2, + usage: wgpu_types::TextureUses::RESOURCE, + range: wgpu_types::ImageSubresourceRange::default(), + }; + let texture_view = unsafe { device.create_texture_view(&texture, &view_desc).unwrap() }; + + let global_group = { + // SAFETY: This is the same size that was specified for buffer creation. + let global_buffer_binding = hal::BufferBinding::new_unchecked( + &global_buffer, + 0, + NonZeroU64::new(global_buffer_desc.size), + ); + let texture_binding = hal::TextureBinding { + view: &texture_view, + usage: wgpu_types::TextureUses::RESOURCE, + }; + let global_group_desc = hal::BindGroupDescriptor { + label: Some("global"), + layout: &global_group_layout, + buffers: &[global_buffer_binding], + samplers: &[&sampler], + textures: &[texture_binding], + acceleration_structures: &[], + external_textures: &[], + entries: &[ + hal::BindGroupEntry { + binding: 0, + resource_index: 0, + count: 1, + }, + hal::BindGroupEntry { + binding: 1, + resource_index: 0, + count: 1, + }, + hal::BindGroupEntry { + binding: 2, + resource_index: 0, + count: 1, + }, + ], + }; + unsafe { device.create_bind_group(&global_group_desc).unwrap() } + }; + + let local_group = { + // SAFETY: The size must fit within the buffer. + let local_buffer_binding = hal::BufferBinding::new_unchecked( + &local_buffer, + 0, + wgpu_types::BufferSize::new(size_of::() as _), + ); + let local_group_desc = hal::BindGroupDescriptor { + label: Some("local"), + layout: &local_group_layout, + buffers: &[local_buffer_binding], + samplers: &[], + textures: &[], + acceleration_structures: &[], + external_textures: &[], + entries: &[hal::BindGroupEntry { + binding: 0, + resource_index: 0, + count: 1, + }], + }; + unsafe { device.create_bind_group(&local_group_desc).unwrap() } + }; + + let init_fence_value = 1; + let fence = unsafe { + let mut fence = device.create_fence().unwrap(); + let init_cmd = cmd_encoder.end_encoding().unwrap(); + queue + .submit(&[&init_cmd], &[], (&mut fence, init_fence_value)) + .unwrap(); + device.wait(&fence, init_fence_value, None).unwrap(); + device.destroy_buffer(staging_buffer); + cmd_encoder.reset_all(iter::once(init_cmd)); + fence + }; + + Ok(Example { + instance, + surface, + surface_format: surface_config.format, + adapter, + device, + queue, + pipeline_layout, + shader, + pipeline, + global_group, + local_group, + global_group_layout, + local_group_layout, + bunnies: Vec::new(), + local_buffer, + local_alignment, + global_buffer, + sampler, + texture, + texture_view, + contexts: vec![ExecutionContext { + encoder: cmd_encoder, + fence, + fence_value: init_fence_value + 1, + used_views: Vec::new(), + used_cmd_bufs: Vec::new(), + frames_recorded: 0, + }], + context_index: 0, + extent: [window_size.0, window_size.1], + start: Instant::now(), + }) + } + + fn is_empty(&self) -> bool { + self.bunnies.is_empty() + } + + fn exit(mut self) { + unsafe { + { + let ctx = &mut self.contexts[self.context_index]; + self.queue + .submit(&[], &[], (&mut ctx.fence, ctx.fence_value)) + .unwrap(); + } + + for mut ctx in self.contexts { + ctx.wait_and_clear(&self.device); + drop(ctx.encoder); + self.device.destroy_fence(ctx.fence); + } + + self.device.destroy_bind_group(self.local_group); + self.device.destroy_bind_group(self.global_group); + self.device.destroy_buffer(self.local_buffer); + self.device.destroy_buffer(self.global_buffer); + self.device.destroy_texture_view(self.texture_view); + self.device.destroy_texture(self.texture); + self.device.destroy_sampler(self.sampler); + self.device.destroy_shader_module(self.shader); + self.device.destroy_render_pipeline(self.pipeline); + self.device + .destroy_bind_group_layout(self.local_group_layout); + self.device + .destroy_bind_group_layout(self.global_group_layout); + self.device.destroy_pipeline_layout(self.pipeline_layout); + + self.surface.unconfigure(&self.device); + drop(self.queue); + drop(self.device); + drop(self.surface); + drop(self.adapter); + } + } + + fn update(&mut self, event: winit::event::WindowEvent) { + if let winit::event::WindowEvent::KeyboardInput { + event: + KeyEvent { + logical_key: Key::Named(NamedKey::Space), + state: ElementState::Pressed, + .. + }, + .. + } = event + { + let spawn_count = 64 + self.bunnies.len() / 2; + let elapsed = self.start.elapsed(); + let color = elapsed.as_nanos() as u32; + println!( + "Spawning {} bunnies, total at {}", + spawn_count, + self.bunnies.len() + spawn_count + ); + for i in 0..spawn_count { + let random = ((elapsed.as_nanos() * (i + 1) as u128) & 0xFF) as f32 / 255.0; + let speed = random * MAX_VELOCITY - (MAX_VELOCITY * 0.5); + self.bunnies.push(Locals { + position: [0.0, 0.5 * (self.extent[1] as f32)], + velocity: [speed, 0.0], + color, + _pad: 0, + }); + } + } + } + + fn render(&mut self) { + let delta = 0.01; + for bunny in self.bunnies.iter_mut() { + bunny.position[0] += bunny.velocity[0] * delta; + bunny.position[1] += bunny.velocity[1] * delta; + bunny.velocity[1] += GRAVITY * delta; + if (bunny.velocity[0] > 0.0 + && bunny.position[0] + 0.5 * BUNNY_SIZE > self.extent[0] as f32) + || (bunny.velocity[0] < 0.0 && bunny.position[0] - 0.5 * BUNNY_SIZE < 0.0) + { + bunny.velocity[0] *= -1.0; + } + if bunny.velocity[1] < 0.0 && bunny.position[1] < 0.5 * BUNNY_SIZE { + bunny.velocity[1] *= -1.0; + } + } + + if !self.bunnies.is_empty() { + let size = self.bunnies.len() * self.local_alignment as usize; + unsafe { + let mapping = self + .device + .map_buffer(&self.local_buffer, 0..size as wgpu_types::BufferAddress) + .unwrap(); + ptr::copy_nonoverlapping( + self.bunnies.as_ptr() as *const u8, + mapping.ptr.as_ptr(), + size, + ); + assert!(mapping.is_coherent); + self.device.unmap_buffer(&self.local_buffer); + } + } + + let ctx = &mut self.contexts[self.context_index]; + + let surface_tex = unsafe { + self.surface + .acquire_texture(None, &ctx.fence) + .unwrap() + .texture + }; + + let target_barrier0 = hal::TextureBarrier { + texture: surface_tex.borrow(), + range: wgpu_types::ImageSubresourceRange::default(), + usage: hal::StateTransition { + from: wgpu_types::TextureUses::UNINITIALIZED, + to: wgpu_types::TextureUses::COLOR_TARGET, + }, + }; + unsafe { + ctx.encoder.begin_encoding(Some("frame")).unwrap(); + ctx.encoder.transition_textures(iter::once(target_barrier0)); + } + + let surface_view_desc = hal::TextureViewDescriptor { + label: None, + format: self.surface_format, + dimension: wgpu_types::TextureViewDimension::D2, + usage: wgpu_types::TextureUses::COLOR_TARGET, + range: wgpu_types::ImageSubresourceRange::default(), + }; + let surface_tex_view = unsafe { + self.device + .create_texture_view(surface_tex.borrow(), &surface_view_desc) + .unwrap() + }; + let pass_desc = hal::RenderPassDescriptor { + label: None, + extent: wgpu_types::Extent3d { + width: self.extent[0], + height: self.extent[1], + depth_or_array_layers: 1, + }, + sample_count: 1, + color_attachments: &[Some(hal::ColorAttachment { + target: hal::Attachment { + view: &surface_tex_view, + usage: wgpu_types::TextureUses::COLOR_TARGET, + }, + depth_slice: None, + resolve_target: None, + ops: hal::AttachmentOps::STORE | hal::AttachmentOps::LOAD_CLEAR, + clear_value: wgpu_types::Color { + r: 0.1, + g: 0.2, + b: 0.3, + a: 1.0, + }, + })], + depth_stencil_attachment: None, + multiview_mask: None, + timestamp_writes: None, + occlusion_query_set: None, + }; + unsafe { + ctx.encoder.begin_render_pass(&pass_desc).unwrap(); + ctx.encoder.set_render_pipeline(&self.pipeline); + ctx.encoder + .set_bind_group(&self.pipeline_layout, 0, &self.global_group, &[]); + } + + for i in 0..self.bunnies.len() { + let offset = (i as wgpu_types::DynamicOffset) + * (self.local_alignment as wgpu_types::DynamicOffset); + unsafe { + ctx.encoder + .set_bind_group(&self.pipeline_layout, 1, &self.local_group, &[offset]); + ctx.encoder.draw(0, 4, 0, 1); + } + } + + ctx.frames_recorded += 1; + + let target_barrier1 = hal::TextureBarrier { + texture: surface_tex.borrow(), + range: wgpu_types::ImageSubresourceRange::default(), + usage: hal::StateTransition { + from: wgpu_types::TextureUses::COLOR_TARGET, + to: wgpu_types::TextureUses::PRESENT, + }, + }; + unsafe { + ctx.encoder.end_render_pass(); + ctx.encoder.transition_textures(iter::once(target_barrier1)); + } + + unsafe { + let cmd_buf = ctx.encoder.end_encoding().unwrap(); + self.queue + .submit( + &[&cmd_buf], + &[&surface_tex], + (&mut ctx.fence, ctx.fence_value), + ) + .unwrap(); + self.queue.present(&self.surface, surface_tex).unwrap(); + ctx.used_cmd_bufs.push(cmd_buf); + ctx.used_views.push(surface_tex_view); + }; + + log::debug!("Context switch from {}", self.context_index); + let old_fence_value = ctx.fence_value; + if self.contexts.len() == 1 { + let hal_desc = hal::CommandEncoderDescriptor { + label: None, + queue: &self.queue, + }; + self.contexts.push(unsafe { + ExecutionContext { + encoder: self.device.create_command_encoder(&hal_desc).unwrap(), + fence: self.device.create_fence().unwrap(), + fence_value: 0, + used_views: Vec::new(), + used_cmd_bufs: Vec::new(), + frames_recorded: 0, + } + }); + } + self.context_index = (self.context_index + 1) % self.contexts.len(); + let next = &mut self.contexts[self.context_index]; + unsafe { + next.wait_and_clear(&self.device); + } + next.fence_value = old_fence_value + 1; + } +} + +cfg_if::cfg_if! { + // Apple + Metal + if #[cfg(all(target_vendor = "apple", feature = "metal"))] { + type Api = hal::api::Metal; + } + // Wasm + Vulkan + else if #[cfg(all(not(target_arch = "wasm32"), feature = "vulkan"))] { + type Api = hal::api::Vulkan; + } + // Windows + DX12 + else if #[cfg(all(windows, feature = "dx12"))] { + type Api = hal::api::Dx12; + } + // Anything + GLES + else if #[cfg(feature = "gles")] { + type Api = hal::api::Gles; + } + // Fallback + else { + type Api = hal::api::Noop; + } +} + +struct App { + example: Option>, + window: Option, + last_frame_inst: Instant, + frame_count: u32, + accum_time: f32, +} + +impl ApplicationHandler for App { + fn resumed(&mut self, event_loop: &ActiveEventLoop) { + if self.window.is_some() { + return; + } + let window = event_loop + .create_window(Window::default_attributes().with_title("hal-bunnymark")) + .unwrap(); + let example = Example::::init(&window).expect("Selected backend is not supported"); + self.window = Some(window); + self.example = Some(example); + println!("Press space to spawn bunnies."); + self.window.as_ref().unwrap().request_redraw(); + } + + fn exiting(&mut self, _event_loop: &ActiveEventLoop) { + self.example.take().unwrap().exit(); + } + + fn window_event( + &mut self, + event_loop: &ActiveEventLoop, + _window_id: winit::window::WindowId, + event: WindowEvent, + ) { + event_loop.set_control_flow(ControlFlow::Poll); + + match event { + WindowEvent::KeyboardInput { + event: + KeyEvent { + logical_key: Key::Named(NamedKey::Escape), + state: ElementState::Pressed, + .. + }, + .. + } + | WindowEvent::CloseRequested => event_loop.exit(), + WindowEvent::RedrawRequested => { + let ex = self.example.as_mut().unwrap(); + { + self.accum_time += self.last_frame_inst.elapsed().as_secs_f32(); + self.last_frame_inst = Instant::now(); + self.frame_count += 1; + if self.frame_count == 100 && !ex.is_empty() { + println!( + "Avg frame time {}ms", + self.accum_time * 1000.0 / self.frame_count as f32 + ); + self.accum_time = 0.0; + self.frame_count = 0; + } + } + ex.render(); + self.window.as_ref().unwrap().request_redraw(); + } + _ => { + self.example.as_mut().unwrap().update(event); + } + } + } +} + +fn main() { + env_logger::init(); + + let event_loop = winit::event_loop::EventLoop::new().unwrap(); + let mut app = App { + example: None, + window: None, + last_frame_inst: Instant::now(), + frame_count: 0, + accum_time: 0.0, + }; + event_loop.run_app(&mut app).unwrap(); +} diff --git a/third_party/wgpu-hal-29.0.4/examples/halmark/shader.wgsl b/third_party/wgpu-hal-29.0.4/examples/halmark/shader.wgsl new file mode 100644 index 0000000..ffa7264 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/examples/halmark/shader.wgsl @@ -0,0 +1,50 @@ +struct Globals { + mvp: mat4x4, + size: vec2, + _pad0: u32, + _pad1: u32, +}; + +struct Locals { + position: vec2, + velocity: vec2, + color: u32, + _pad0: u32, + _pad1: u32, + _pad2: u32, +}; + +@group(0) +@binding(0) +var globals: Globals; + +@group(1) +@binding(0) +var locals: Locals; + +struct VertexOutput { + @builtin(position) position: vec4, + @location(0) tex_coords: vec2, + @location(1) color: vec4, +}; + +@vertex +fn vs_main(@builtin(vertex_index) vi: u32) -> VertexOutput { + let tc = vec2(f32(vi & 1u), 0.5 * f32(vi & 2u)); + let offset = vec2(tc.x * globals.size.x, tc.y * globals.size.y); + let pos = globals.mvp * vec4(locals.position + offset, 0.0, 1.0); + let color = vec4((vec4(locals.color) >> vec4(0u, 8u, 16u, 24u)) & vec4(255u)) / 255.0; + return VertexOutput(pos, tc, color); +} + +@group(0) +@binding(1) +var tex: texture_2d; +@group(0) +@binding(2) +var sam: sampler; + +@fragment +fn fs_main(vertex: VertexOutput) -> @location(0) vec4 { + return vertex.color * textureSampleLevel(tex, sam, vertex.tex_coords, 0.0); +} diff --git a/third_party/wgpu-hal-29.0.4/examples/raw-gles.em.html b/third_party/wgpu-hal-29.0.4/examples/raw-gles.em.html new file mode 100644 index 0000000..f3587e8 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/examples/raw-gles.em.html @@ -0,0 +1,16 @@ + + + + + + + + + + + \ No newline at end of file diff --git a/third_party/wgpu-hal-29.0.4/examples/raw-gles.rs b/third_party/wgpu-hal-29.0.4/examples/raw-gles.rs new file mode 100644 index 0000000..b32398c --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/examples/raw-gles.rs @@ -0,0 +1,396 @@ +//! This example shows interop with raw GLES contexts - +//! the ability to hook up wgpu-hal to an existing context and draw into it. +//! +//! Emscripten build: +//! 1. install emsdk +//! 2. build this example with cargo: +//! EMCC_CFLAGS="-g -s ERROR_ON_UNDEFINED_SYMBOLS=0 --no-entry -s FULL_ES3=1" cargo build --example raw-gles --target wasm32-unknown-emscripten +//! 3. copy raw-gles.em.html into target directory and open it in browser: +//! cp wgpu-hal/examples/raw-gles.em.html target/wasm32-unknown-emscripten/debug/examples + +extern crate wgpu_hal as hal; + +#[cfg(not(any( + target_arch = "wasm32", + target_os = "ios", + target_os = "visionos", + target_env = "ohos" +)))] +fn main() { + use std::ffi::CString; + + use glutin::{ + config::GlConfig as _, + context::{NotCurrentGlContext as _, PossiblyCurrentGlContext as _, Version}, + display::{GetGlDisplay as _, GlDisplay as _}, + surface::GlSurface as _, + }; + use glutin_winit::GlWindow as _; + use raw_window_handle::HasWindowHandle as _; + use winit::{ + application::ApplicationHandler, + event::{KeyEvent, WindowEvent}, + event_loop::{ActiveEventLoop, ControlFlow, EventLoop}, + keyboard::{Key, NamedKey}, + window::Window, + }; + + env_logger::init(); + println!("Initializing external GL context"); + + /// Find the config with the maximum number of samples, so our triangle will be + /// smooth. + fn gl_config_picker( + configs: Box + '_>, + ) -> glutin::config::Config { + configs + .reduce(|accum, config| { + if config.num_samples() > accum.num_samples() { + config + } else { + accum + } + }) + .expect("Failed to find a matching config") + } + + struct App { + gl_config: Option, + not_current_gl_context: Option, + state: Option<( + glutin::context::PossiblyCurrentContext, + glutin::surface::Surface, + Window, + )>, + exposed: Option>, + } + + impl App { + fn new() -> Self { + Self { + gl_config: None, + not_current_gl_context: None, + state: None, + exposed: None, + } + } + } + + impl ApplicationHandler for App { + fn resumed(&mut self, event_loop: &ActiveEventLoop) { + // First call: create display, pick config, create GL context. + if self.gl_config.is_none() { + // Only Windows requires the window to be present before creating the display. + // Other platforms don't really need one. + let window_attributes = cfg!(windows).then(|| { + Window::default_attributes() + .with_title("wgpu raw GLES example (press Escape to exit)") + }); + + let template = glutin::config::ConfigTemplateBuilder::new(); + + let display_builder = + glutin_winit::DisplayBuilder::new().with_window_attributes(window_attributes); + + let (window, gl_config) = display_builder + .build(event_loop, template, gl_config_picker) + .expect("Failed to build window and config from display"); + + println!("Picked a config with {} samples", gl_config.num_samples()); + + let raw_window_handle = window + .as_ref() + .and_then(|window| window.window_handle().ok()) + .map(|handle| handle.as_raw()); + + let gl_display = gl_config.display(); + + // Glutin tries to create an OpenGL context by default. + // Force it to use any version of GLES. + let context_attributes = glutin::context::ContextAttributesBuilder::new() + // wgpu expects GLES 3.0+. + .with_context_api(glutin::context::ContextApi::Gles(Some(Version::new(3, 0)))) + .build(raw_window_handle); + + let gl_context = unsafe { + gl_display + .create_context(&gl_config, &context_attributes) + .expect("failed to create context") + }; + + self.not_current_gl_context = Some(gl_context); + self.gl_config = Some(gl_config); + + // On Windows, the window was already created by DisplayBuilder. + if let Some(window) = window { + self.create_surface(event_loop, window); + return; + } + } + + // Create window + surface (non-Windows first call, or Android re-resume). + if self.state.is_none() { + let gl_config = self.gl_config.as_ref().unwrap(); + let window = glutin_winit::finalize_window( + event_loop, + Window::default_attributes() + .with_title("wgpu raw GLES example (press Escape to exit)"), + gl_config, + ) + .unwrap(); + self.create_surface(event_loop, window); + } + } + + fn suspended(&mut self, _event_loop: &ActiveEventLoop) { + // This event is only raised on Android, where the backing NativeWindow for a GL + // Surface can appear and disappear at any moment. + println!("Android window removed"); + + // Destroy the GL Surface and un-current the GL Context before ndk-glue releases + // the window back to the system. + if let Some((gl_context, ..)) = self.state.take() { + assert!(self + .not_current_gl_context + .replace(gl_context.make_not_current().unwrap()) + .is_none()); + } + } + + fn window_event( + &mut self, + event_loop: &ActiveEventLoop, + _window_id: winit::window::WindowId, + event: WindowEvent, + ) { + event_loop.set_control_flow(ControlFlow::Wait); + + match event { + WindowEvent::CloseRequested + | WindowEvent::KeyboardInput { + event: + KeyEvent { + logical_key: Key::Named(NamedKey::Escape), + .. + }, + .. + } => event_loop.exit(), + WindowEvent::Resized(size) => { + if size.width != 0 && size.height != 0 { + // Some platforms like EGL require resizing GL surface to update the size. + // Notable platforms here are Wayland and macOS, other don't require it + // and the function is no-op, but it's wise to resize it for portability + // reasons. + if let Some((gl_context, gl_surface, window)) = &self.state { + window.resize_surface(gl_surface, gl_context); + } + } + } + WindowEvent::RedrawRequested => { + if let (Some(exposed), Some((gl_context, gl_surface, window))) = + (&self.exposed, &self.state) + { + let inner_size = window.inner_size(); + + fill_screen(exposed, inner_size.width, inner_size.height); + + println!("Showing the window"); + gl_surface + .swap_buffers(gl_context) + .expect("Failed to swap buffers"); + } + } + _ => (), + } + } + } + + impl App { + fn create_surface(&mut self, _event_loop: &ActiveEventLoop, window: Window) { + let gl_config = self.gl_config.as_ref().unwrap(); + + let attrs = window + .build_surface_attributes(Default::default()) + .expect("Failed to build surface attributes"); + let gl_surface = unsafe { + gl_config + .display() + .create_window_surface(gl_config, &attrs) + .expect("Cannot create GL WindowSurface") + }; + + // Make it current. + let gl_context = self + .not_current_gl_context + .take() + .unwrap() + .make_current(&gl_surface) + .expect("GL context cannot be made current with WindowSurface"); + + // The context needs to be current for the Renderer to set up shaders and + // buffers. It also performs function loading, which needs a current context on + // WGL. + println!("Hooking up to wgpu-hal"); + self.exposed.get_or_insert_with(|| { + unsafe { + ::Adapter::new_external( + |name| { + // XXX: On WGL this should only be called after the context was + // made current + gl_config + .display() + .get_proc_address(&CString::new(name).expect(name)) + }, + wgpu_types::GlBackendOptions::default(), + ) + } + .expect("GL adapter can't be initialized") + }); + + window.request_redraw(); + self.state = Some((gl_context, gl_surface, window)); + } + } + + let event_loop = EventLoop::new().unwrap(); + let mut app = App::new(); + event_loop + .run_app(&mut app) + .expect("Couldn't run event loop"); +} + +#[cfg(target_os = "emscripten")] +fn main() { + env_logger::init(); + + println!("Initializing external GL context"); + let egl = khronos_egl::Instance::new(khronos_egl::Static); + let display = unsafe { egl.get_display(khronos_egl::DEFAULT_DISPLAY) }.unwrap(); + egl.initialize(display) + .expect("unable to initialize display"); + + let attributes = [ + khronos_egl::RED_SIZE, + 8, + khronos_egl::GREEN_SIZE, + 8, + khronos_egl::BLUE_SIZE, + 8, + khronos_egl::NONE, + ]; + + let config = egl + .choose_first_config(display, &attributes) + .unwrap() + .expect("unable to choose config"); + let surface = unsafe { + let window = std::ptr::null_mut::(); + egl.create_window_surface(display, config, window, None) + } + .expect("unable to create surface"); + + let context_attributes = [khronos_egl::CONTEXT_CLIENT_VERSION, 3, khronos_egl::NONE]; + + let gl_context = egl + .create_context(display, config, None, &context_attributes) + .expect("unable to create context"); + egl.make_current(display, Some(surface), Some(surface), Some(gl_context)) + .expect("can't make context current"); + + println!("Hooking up to wgpu-hal"); + let exposed = unsafe { + ::Adapter::new_external(|name| { + egl.get_proc_address(name) + .map_or(std::ptr::null(), |p| p as *const _) + }) + } + .expect("GL adapter can't be initialized"); + + fill_screen(&exposed, 640, 400); +} + +#[cfg(any( + all(target_arch = "wasm32", not(target_os = "emscripten")), + target_os = "ios", + target_os = "visionos", + target_env = "ohos" +))] +fn main() { + eprintln!("This example is not supported on this platform") +} + +#[cfg(not(any( + all(target_arch = "wasm32", not(target_os = "emscripten")), + target_os = "ios", + target_os = "visionos" +)))] +fn fill_screen(exposed: &hal::ExposedAdapter, width: u32, height: u32) { + use hal::{Adapter as _, CommandEncoder as _, Device as _, Queue as _}; + + let od = unsafe { + exposed.adapter.open( + wgpu_types::Features::empty(), + &wgpu_types::Limits::downlevel_defaults(), + &wgpu_types::MemoryHints::default(), + ) + } + .unwrap(); + + let format = wgpu_types::TextureFormat::Rgba8UnormSrgb; + let texture = ::Texture::default_framebuffer(format); + let view = unsafe { + od.device + .create_texture_view( + &texture, + &hal::TextureViewDescriptor { + label: None, + format, + dimension: wgpu_types::TextureViewDimension::D2, + usage: wgpu_types::TextureUses::COLOR_TARGET, + range: wgpu_types::ImageSubresourceRange::default(), + }, + ) + .unwrap() + }; + + println!("Filling the screen"); + let mut encoder = unsafe { + od.device + .create_command_encoder(&hal::CommandEncoderDescriptor { + label: None, + queue: &od.queue, + }) + .unwrap() + }; + let mut fence = unsafe { od.device.create_fence().unwrap() }; + let rp_desc = hal::RenderPassDescriptor { + label: None, + extent: wgpu_types::Extent3d { + width, + height, + depth_or_array_layers: 1, + }, + sample_count: 1, + color_attachments: &[Some(hal::ColorAttachment { + target: hal::Attachment { + view: &view, + usage: wgpu_types::TextureUses::COLOR_TARGET, + }, + depth_slice: None, + resolve_target: None, + ops: hal::AttachmentOps::STORE | hal::AttachmentOps::LOAD_CLEAR, + clear_value: wgpu_types::Color::BLUE, + })], + depth_stencil_attachment: None, + multiview_mask: None, + timestamp_writes: None, + occlusion_query_set: None, + }; + unsafe { + encoder.begin_encoding(None).unwrap(); + encoder.begin_render_pass(&rp_desc).unwrap(); + encoder.end_render_pass(); + let cmd_buf = encoder.end_encoding().unwrap(); + od.queue.submit(&[&cmd_buf], &[], (&mut fence, 0)).unwrap(); + } +} diff --git a/third_party/wgpu-hal-29.0.4/examples/ray-traced-triangle/main.rs b/third_party/wgpu-hal-29.0.4/examples/ray-traced-triangle/main.rs new file mode 100644 index 0000000..b486520 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/examples/ray-traced-triangle/main.rs @@ -0,0 +1,1223 @@ +extern crate wgpu_hal as hal; + +use hal::{ + Adapter as _, CommandEncoder as _, Device as _, Instance as _, Queue as _, Surface as _, +}; +use raw_window_handle::{HasDisplayHandle, HasWindowHandle}; + +use glam::{Affine3A, Mat4, Vec3}; +use std::{ + borrow::{Borrow, Cow}, + iter, ptr, + time::Instant, +}; +use wgpu_types::Dx12BackendOptions; +use winit::{ + application::ApplicationHandler, + event_loop::{ActiveEventLoop, ControlFlow}, + window::{Window, WindowButtons}, +}; + +const DESIRED_MAX_LATENCY: u32 = 2; + +/// [D3D12_RAYTRACING_INSTANCE_DESC](https://microsoft.github.io/DirectX-Specs/d3d/Raytracing.html#d3d12_raytracing_instance_desc) +/// [VkAccelerationStructureInstanceKHR](https://registry.khronos.org/vulkan/specs/1.3-extensions/man/html/VkAccelerationStructureInstanceKHR.html) +#[derive(Clone)] +#[repr(C)] +struct AccelerationStructureInstance { + transform: [f32; 12], + custom_index_and_mask: u32, + shader_binding_table_record_offset_and_flags: u32, + acceleration_structure_reference: u64, +} + +impl std::fmt::Debug for AccelerationStructureInstance { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("Instance") + .field("transform", &self.transform) + .field("custom_data()", &self.custom_index()) + .field("mask()", &self.mask()) + .field( + "shader_binding_table_record_offset()", + &self.shader_binding_table_record_offset(), + ) + .field("flags()", &self.flags()) + .field( + "acceleration_structure_reference", + &self.acceleration_structure_reference, + ) + .finish() + } +} + +#[allow(dead_code)] +impl AccelerationStructureInstance { + const LOW_24_MASK: u32 = 0x00ff_ffff; + const MAX_U24: u32 = (1u32 << 24u32) - 1u32; + + #[inline] + fn affine_to_rows(mat: &Affine3A) -> [f32; 12] { + let row_0 = mat.matrix3.row(0); + let row_1 = mat.matrix3.row(1); + let row_2 = mat.matrix3.row(2); + let translation = mat.translation; + [ + row_0.x, + row_0.y, + row_0.z, + translation.x, + row_1.x, + row_1.y, + row_1.z, + translation.y, + row_2.x, + row_2.y, + row_2.z, + translation.z, + ] + } + + #[inline] + fn rows_to_affine(rows: &[f32; 12]) -> Affine3A { + Affine3A::from_cols_array(&[ + rows[0], rows[3], rows[6], rows[9], rows[1], rows[4], rows[7], rows[10], rows[2], + rows[5], rows[8], rows[11], + ]) + } + + pub fn transform_as_affine(&self) -> Affine3A { + Self::rows_to_affine(&self.transform) + } + pub fn set_transform(&mut self, transform: &Affine3A) { + self.transform = Self::affine_to_rows(transform); + } + + pub fn custom_index(&self) -> u32 { + self.custom_index_and_mask & Self::LOW_24_MASK + } + + pub fn mask(&self) -> u8 { + (self.custom_index_and_mask >> 24) as u8 + } + + pub fn shader_binding_table_record_offset(&self) -> u32 { + self.shader_binding_table_record_offset_and_flags & Self::LOW_24_MASK + } + + pub fn flags(&self) -> u8 { + (self.shader_binding_table_record_offset_and_flags >> 24) as u8 + } + + pub fn set_custom_index(&mut self, custom_index: u32) { + debug_assert!( + custom_index <= Self::MAX_U24, + "custom_index uses more than 24 bits! {custom_index} > {}", + Self::MAX_U24 + ); + self.custom_index_and_mask = + (custom_index & Self::LOW_24_MASK) | (self.custom_index_and_mask & !Self::LOW_24_MASK) + } + + pub fn set_mask(&mut self, mask: u8) { + self.custom_index_and_mask = + (self.custom_index_and_mask & Self::LOW_24_MASK) | (u32::from(mask) << 24) + } + + pub fn set_shader_binding_table_record_offset( + &mut self, + shader_binding_table_record_offset: u32, + ) { + debug_assert!( + shader_binding_table_record_offset <= Self::MAX_U24, + "shader_binding_table_record_offset uses more than 24 bits! {} > {}", + shader_binding_table_record_offset, + Self::MAX_U24 + ); + self.shader_binding_table_record_offset_and_flags = (shader_binding_table_record_offset + & Self::LOW_24_MASK) + | (self.shader_binding_table_record_offset_and_flags & !Self::LOW_24_MASK) + } + + pub fn set_flags(&mut self, flags: u8) { + self.shader_binding_table_record_offset_and_flags = + (self.shader_binding_table_record_offset_and_flags & Self::LOW_24_MASK) + | (u32::from(flags) << 24) + } + + pub fn new( + transform: &Affine3A, + custom_index: u32, + mask: u8, + shader_binding_table_record_offset: u32, + flags: u8, + acceleration_structure_reference: u64, + ) -> Self { + debug_assert!( + custom_index <= Self::MAX_U24, + "custom_index uses more than 24 bits! {custom_index} > {}", + Self::MAX_U24 + ); + debug_assert!( + shader_binding_table_record_offset <= Self::MAX_U24, + "shader_binding_table_record_offset uses more than 24 bits! {} > {}", + shader_binding_table_record_offset, + Self::MAX_U24 + ); + AccelerationStructureInstance { + transform: Self::affine_to_rows(transform), + custom_index_and_mask: (custom_index & Self::MAX_U24) | (u32::from(mask) << 24), + shader_binding_table_record_offset_and_flags: (shader_binding_table_record_offset + & Self::MAX_U24) + | (u32::from(flags) << 24), + acceleration_structure_reference, + } + } +} + +struct ExecutionContext { + encoder: A::CommandEncoder, + fence: A::Fence, + fence_value: hal::FenceValue, + used_views: Vec, + used_cmd_bufs: Vec, + frames_recorded: usize, +} + +impl ExecutionContext { + unsafe fn wait_and_clear(&mut self, device: &A::Device) { + device.wait(&self.fence, self.fence_value, None).unwrap(); + self.encoder.reset_all(self.used_cmd_bufs.drain(..)); + for view in self.used_views.drain(..) { + device.destroy_texture_view(view); + } + self.frames_recorded = 0; + } +} + +#[allow(dead_code)] +struct Example { + instance: A::Instance, + adapter: A::Adapter, + surface: A::Surface, + surface_format: wgpu_types::TextureFormat, + device: A::Device, + queue: A::Queue, + + contexts: Vec>, + context_index: usize, + extent: [u32; 2], + start: Instant, + pipeline: A::ComputePipeline, + bind_group: A::BindGroup, + bgl: A::BindGroupLayout, + shader_module: A::ShaderModule, + texture_view: A::TextureView, + uniform_buffer: A::Buffer, + pipeline_layout: A::PipelineLayout, + vertices_buffer: A::Buffer, + indices_buffer: Option, + texture: A::Texture, + instances: [AccelerationStructureInstance; 3], + instances_buffer: A::Buffer, + blas: A::AccelerationStructure, + tlas: A::AccelerationStructure, + scratch_buffer: A::Buffer, + time: f32, +} + +impl Example { + fn init(window: &winit::window::Window) -> Result> { + let mut index_buffer = false; + + for arg in std::env::args() { + if arg == "index_buffer" { + index_buffer = true; + } + } + + if index_buffer { + log::info!("using index buffer") + } + + // The Instance can be initialized with the DisplayHandle from the EventLoop as well + let raw_display_handle = window.display_handle()?; + + let instance_desc = hal::InstanceDescriptor { + name: "example", + flags: wgpu_types::InstanceFlags::default(), + memory_budget_thresholds: wgpu_types::MemoryBudgetThresholds::default(), + backend_options: wgpu_types::BackendOptions { + dx12: Dx12BackendOptions { + shader_compiler: wgpu_types::Dx12Compiler::default_dynamic_dxc(), + ..Default::default() + }, + ..Default::default() + }, + telemetry: None, + display: Some(raw_display_handle), + }; + let instance = unsafe { A::Instance::init(&instance_desc)? }; + let surface = { + let raw_window_handle = window.window_handle()?.as_raw(); + + unsafe { + instance + .create_surface(raw_display_handle.as_raw(), raw_window_handle) + .unwrap() + } + }; + + let (adapter, features) = unsafe { + let mut adapters = instance.enumerate_adapters(Some(&surface)); + if adapters.is_empty() { + panic!("No adapters found"); + } + let exposed = adapters.swap_remove(0); + dbg!(exposed.features); + (exposed.adapter, exposed.features) + }; + let surface_caps = unsafe { adapter.surface_capabilities(&surface) } + .expect("Surface doesn't support presentation"); + log::info!("Surface caps: {surface_caps:#?}"); + + let hal::OpenDevice { device, queue } = unsafe { + adapter + .open( + features, + &wgpu_types::Limits::default(), + &wgpu_types::MemoryHints::Performance, + ) + .unwrap() + }; + + let window_size: (u32, u32) = window.inner_size().into(); + dbg!(&surface_caps.formats); + let surface_format = if surface_caps + .formats + .contains(&wgpu_types::TextureFormat::Rgba8Unorm) + { + wgpu_types::TextureFormat::Rgba8Unorm + } else { + *surface_caps.formats.first().unwrap() + }; + let surface_config = hal::SurfaceConfiguration { + maximum_frame_latency: DESIRED_MAX_LATENCY + .max(*surface_caps.maximum_frame_latency.start()) + .min(*surface_caps.maximum_frame_latency.end()), + present_mode: wgpu_types::PresentMode::Fifo, + composite_alpha_mode: wgpu_types::CompositeAlphaMode::Opaque, + format: surface_format, + extent: wgpu_types::Extent3d { + width: window_size.0, + height: window_size.1, + depth_or_array_layers: 1, + }, + usage: wgpu_types::TextureUses::COLOR_TARGET | wgpu_types::TextureUses::COPY_DST, + view_formats: vec![surface_format], + }; + unsafe { + surface.configure(&device, &surface_config).unwrap(); + }; + + #[allow(dead_code)] + struct Uniforms { + view_inverse: glam::Mat4, + proj_inverse: glam::Mat4, + } + + let bgl_desc = hal::BindGroupLayoutDescriptor { + label: None, + flags: hal::BindGroupLayoutFlags::empty(), + entries: &[ + wgpu_types::BindGroupLayoutEntry { + binding: 0, + visibility: wgpu_types::ShaderStages::COMPUTE, + ty: wgpu_types::BindingType::Buffer { + ty: wgpu_types::BufferBindingType::Uniform, + has_dynamic_offset: false, + min_binding_size: wgpu_types::BufferSize::new(size_of::() as _), + }, + count: None, + }, + wgpu_types::BindGroupLayoutEntry { + binding: 1, + visibility: wgpu_types::ShaderStages::COMPUTE, + ty: wgpu_types::BindingType::StorageTexture { + access: wgpu_types::StorageTextureAccess::WriteOnly, + format: wgpu_types::TextureFormat::Rgba8Unorm, + view_dimension: wgpu_types::TextureViewDimension::D2, + }, + count: None, + }, + wgpu_types::BindGroupLayoutEntry { + binding: 2, + visibility: wgpu_types::ShaderStages::COMPUTE, + ty: wgpu_types::BindingType::AccelerationStructure { + vertex_return: false, + }, + count: None, + }, + ], + }; + + let bgl = unsafe { device.create_bind_group_layout(&bgl_desc).unwrap() }; + + let naga_shader = { + let shader_file = std::path::PathBuf::from(env!("CARGO_MANIFEST_DIR")) + .join("examples") + .join("ray-traced-triangle") + .join("shader.wgsl"); + let source = std::fs::read_to_string(shader_file).unwrap(); + let module = naga::front::wgsl::Frontend::new().parse(&source).unwrap(); + let info = naga::valid::Validator::new( + naga::valid::ValidationFlags::all(), + naga::valid::Capabilities::RAY_QUERY, + ) + .validate(&module) + .unwrap(); + hal::NagaShader { + module: Cow::Owned(module), + info, + debug_source: None, + } + }; + let shader_desc = hal::ShaderModuleDescriptor { + label: None, + runtime_checks: wgpu_types::ShaderRuntimeChecks::checked(), + }; + let shader_module = unsafe { + device + .create_shader_module(&shader_desc, hal::ShaderInput::Naga(naga_shader)) + .unwrap() + }; + + let pipeline_layout_desc = hal::PipelineLayoutDescriptor { + label: None, + flags: hal::PipelineLayoutFlags::empty(), + bind_group_layouts: &[Some(&bgl)], + immediate_size: 0, + }; + let pipeline_layout = unsafe { + device + .create_pipeline_layout(&pipeline_layout_desc) + .unwrap() + }; + + let pipeline = unsafe { + device.create_compute_pipeline(&hal::ComputePipelineDescriptor { + label: Some("pipeline"), + layout: &pipeline_layout, + stage: hal::ProgrammableStage { + module: &shader_module, + entry_point: "main", + constants: &Default::default(), + zero_initialize_workgroup_memory: true, + }, + cache: None, + }) + } + .unwrap(); + + let vertices: [f32; 9] = [1.0, 1.0, 0.0, -1.0, 1.0, 0.0, 0.0, -1.0, 0.0]; + + let vertices_size_in_bytes = vertices.len() * 4; + + let indices: [u32; 3] = [0, 1, 2]; + + let indices_size_in_bytes = indices.len() * 4; + + let vertices_buffer = unsafe { + let vertices_buffer = device + .create_buffer(&hal::BufferDescriptor { + label: Some("vertices buffer"), + size: vertices_size_in_bytes as u64, + usage: wgpu_types::BufferUses::MAP_WRITE + | wgpu_types::BufferUses::BOTTOM_LEVEL_ACCELERATION_STRUCTURE_INPUT, + memory_flags: hal::MemoryFlags::TRANSIENT | hal::MemoryFlags::PREFER_COHERENT, + }) + .unwrap(); + + let mapping = device + .map_buffer(&vertices_buffer, 0..vertices_size_in_bytes as u64) + .unwrap(); + ptr::copy_nonoverlapping( + vertices.as_ptr() as *const u8, + mapping.ptr.as_ptr(), + vertices_size_in_bytes, + ); + device.unmap_buffer(&vertices_buffer); + assert!(mapping.is_coherent); + + vertices_buffer + }; + + let indices_buffer = if index_buffer { + unsafe { + let indices_buffer = device + .create_buffer(&hal::BufferDescriptor { + label: Some("indices buffer"), + size: indices_size_in_bytes as u64, + usage: wgpu_types::BufferUses::MAP_WRITE + | wgpu_types::BufferUses::BOTTOM_LEVEL_ACCELERATION_STRUCTURE_INPUT, + memory_flags: hal::MemoryFlags::TRANSIENT + | hal::MemoryFlags::PREFER_COHERENT, + }) + .unwrap(); + + let mapping = device + .map_buffer(&indices_buffer, 0..indices_size_in_bytes as u64) + .unwrap(); + ptr::copy_nonoverlapping( + indices.as_ptr() as *const u8, + mapping.ptr.as_ptr(), + indices_size_in_bytes, + ); + device.unmap_buffer(&indices_buffer); + assert!(mapping.is_coherent); + + Some((indices_buffer, indices.len())) + } + } else { + None + }; + + let blas_triangles = vec![hal::AccelerationStructureTriangles { + vertex_buffer: Some(&vertices_buffer), + first_vertex: 0, + vertex_format: wgpu_types::VertexFormat::Float32x3, + // each vertex is 3 floats, and floats are stored raw in the array + vertex_count: vertices.len() as u32 / 3, + vertex_stride: 3 * 4, + indices: indices_buffer.as_ref().map(|(buf, len)| { + hal::AccelerationStructureTriangleIndices { + buffer: Some(buf), + format: wgpu_types::IndexFormat::Uint32, + offset: 0, + count: *len as u32, + } + }), + + transform: None, + flags: hal::AccelerationStructureGeometryFlags::OPAQUE, + }]; + let blas_entries = hal::AccelerationStructureEntries::Triangles(blas_triangles); + + let blas_sizes = unsafe { + device.get_acceleration_structure_build_sizes( + &hal::GetAccelerationStructureBuildSizesDescriptor { + entries: &blas_entries, + flags: hal::AccelerationStructureBuildFlags::PREFER_FAST_TRACE, + }, + ) + }; + + let blas = unsafe { + device.create_acceleration_structure(&hal::AccelerationStructureDescriptor { + label: Some("blas"), + size: blas_sizes.acceleration_structure_size, + format: hal::AccelerationStructureFormat::BottomLevel, + allow_compaction: false, + }) + } + .unwrap(); + + let instances = [ + AccelerationStructureInstance::new( + &Affine3A::from_translation(Vec3 { + x: 0.0, + y: 0.0, + z: 0.0, + }), + 0, + 0xff, + 0, + 0, + unsafe { device.get_acceleration_structure_device_address(&blas) }, + ), + AccelerationStructureInstance::new( + &Affine3A::from_translation(Vec3 { + x: -1.0, + y: -1.0, + z: -2.0, + }), + 0, + 0xff, + 0, + 0, + unsafe { device.get_acceleration_structure_device_address(&blas) }, + ), + AccelerationStructureInstance::new( + &Affine3A::from_translation(Vec3 { + x: 1.0, + y: -1.0, + z: -2.0, + }), + 0, + 0xff, + 0, + 0, + unsafe { device.get_acceleration_structure_device_address(&blas) }, + ), + ]; + + let instances_buffer_size = instances.len() * size_of::(); + + let instances_buffer = unsafe { + let instances_buffer = device + .create_buffer(&hal::BufferDescriptor { + label: Some("instances_buffer"), + size: instances_buffer_size as u64, + usage: wgpu_types::BufferUses::MAP_WRITE + | wgpu_types::BufferUses::TOP_LEVEL_ACCELERATION_STRUCTURE_INPUT, + memory_flags: hal::MemoryFlags::TRANSIENT | hal::MemoryFlags::PREFER_COHERENT, + }) + .unwrap(); + + let mapping = device + .map_buffer(&instances_buffer, 0..instances_buffer_size as u64) + .unwrap(); + ptr::copy_nonoverlapping( + instances.as_ptr() as *const u8, + mapping.ptr.as_ptr(), + instances_buffer_size, + ); + device.unmap_buffer(&instances_buffer); + assert!(mapping.is_coherent); + + instances_buffer + }; + + let mut tlas_entries = + hal::AccelerationStructureEntries::Instances(hal::AccelerationStructureInstances { + buffer: Some(&instances_buffer), + count: 3, + offset: 0, + }); + + let tlas_flags = hal::AccelerationStructureBuildFlags::PREFER_FAST_TRACE + | hal::AccelerationStructureBuildFlags::ALLOW_UPDATE; + + let tlas_sizes = unsafe { + device.get_acceleration_structure_build_sizes( + &hal::GetAccelerationStructureBuildSizesDescriptor { + entries: &tlas_entries, + flags: tlas_flags, + }, + ) + }; + + let tlas = unsafe { + device.create_acceleration_structure(&hal::AccelerationStructureDescriptor { + label: Some("tlas"), + size: tlas_sizes.acceleration_structure_size, + format: hal::AccelerationStructureFormat::TopLevel, + allow_compaction: false, + }) + } + .unwrap(); + + let uniforms = { + let view = Mat4::look_at_rh(Vec3::new(0.0, 0.0, 2.5), Vec3::ZERO, Vec3::Y); + let proj = Mat4::perspective_rh(59.0_f32.to_radians(), 1.0, 0.001, 1000.0); + + Uniforms { + view_inverse: view.inverse(), + proj_inverse: proj.inverse(), + } + }; + + let uniforms_size = size_of::(); + + let uniform_buffer = unsafe { + let uniform_buffer = device + .create_buffer(&hal::BufferDescriptor { + label: Some("uniform buffer"), + size: uniforms_size as u64, + usage: wgpu_types::BufferUses::MAP_WRITE | wgpu_types::BufferUses::UNIFORM, + memory_flags: hal::MemoryFlags::PREFER_COHERENT, + }) + .unwrap(); + + let mapping = device + .map_buffer(&uniform_buffer, 0..uniforms_size as u64) + .unwrap(); + ptr::copy_nonoverlapping( + &uniforms as *const Uniforms as *const u8, + mapping.ptr.as_ptr(), + uniforms_size, + ); + device.unmap_buffer(&uniform_buffer); + assert!(mapping.is_coherent); + uniform_buffer + }; + + let texture_desc = hal::TextureDescriptor { + label: None, + size: wgpu_types::Extent3d { + width: 512, + height: 512, + depth_or_array_layers: 1, + }, + mip_level_count: 1, + sample_count: 1, + dimension: wgpu_types::TextureDimension::D2, + format: wgpu_types::TextureFormat::Rgba8Unorm, + usage: wgpu_types::TextureUses::STORAGE_READ_WRITE | wgpu_types::TextureUses::COPY_SRC, + memory_flags: hal::MemoryFlags::empty(), + view_formats: vec![wgpu_types::TextureFormat::Rgba8Unorm], + }; + let texture = unsafe { device.create_texture(&texture_desc).unwrap() }; + + let view_desc = hal::TextureViewDescriptor { + label: None, + format: texture_desc.format, + dimension: wgpu_types::TextureViewDimension::D2, + usage: wgpu_types::TextureUses::STORAGE_READ_WRITE | wgpu_types::TextureUses::COPY_SRC, + range: wgpu_types::ImageSubresourceRange::default(), + }; + let texture_view = unsafe { device.create_texture_view(&texture, &view_desc).unwrap() }; + + let bind_group = { + let buffer_binding = unsafe { + // SAFETY: The size matches the buffer allocation. + hal::BufferBinding::new_unchecked( + &uniform_buffer, + 0, + wgpu_types::BufferSize::new_unchecked(uniforms_size as u64), + ) + }; + let texture_binding = hal::TextureBinding { + view: &texture_view, + usage: wgpu_types::TextureUses::STORAGE_READ_WRITE, + }; + let group_desc = hal::BindGroupDescriptor { + label: Some("bind group"), + layout: &bgl, + buffers: &[buffer_binding], + samplers: &[], + textures: &[texture_binding], + acceleration_structures: &[&tlas], + external_textures: &[], + entries: &[ + hal::BindGroupEntry { + binding: 0, + resource_index: 0, + count: 1, + }, + hal::BindGroupEntry { + binding: 1, + resource_index: 0, + count: 1, + }, + hal::BindGroupEntry { + binding: 2, + resource_index: 0, + count: 1, + }, + ], + }; + unsafe { device.create_bind_group(&group_desc).unwrap() } + }; + + let scratch_buffer = unsafe { + device + .create_buffer(&hal::BufferDescriptor { + label: Some("scratch buffer"), + size: blas_sizes + .build_scratch_size + .max(tlas_sizes.build_scratch_size), + usage: wgpu_types::BufferUses::ACCELERATION_STRUCTURE_SCRATCH, + memory_flags: hal::MemoryFlags::empty(), + }) + .unwrap() + }; + + if let hal::AccelerationStructureEntries::Instances(ref mut i) = tlas_entries { + assert!( + instances.len() <= i.count as usize, + "Tlas allocation to small" + ); + } + + let cmd_encoder_desc = hal::CommandEncoderDescriptor { + label: None, + queue: &queue, + }; + let mut cmd_encoder = unsafe { device.create_command_encoder(&cmd_encoder_desc).unwrap() }; + + unsafe { cmd_encoder.begin_encoding(Some("init")).unwrap() }; + + unsafe { + cmd_encoder.place_acceleration_structure_barrier(hal::AccelerationStructureBarrier { + usage: hal::StateTransition { + from: hal::AccelerationStructureUses::empty(), + to: hal::AccelerationStructureUses::BUILD_OUTPUT, + }, + }); + + cmd_encoder.build_acceleration_structures( + 1, + [hal::BuildAccelerationStructureDescriptor { + mode: hal::AccelerationStructureBuildMode::Build, + flags: hal::AccelerationStructureBuildFlags::PREFER_FAST_TRACE, + destination_acceleration_structure: &blas, + scratch_buffer: &scratch_buffer, + entries: &blas_entries, + source_acceleration_structure: None, + scratch_buffer_offset: 0, + }], + ); + + let scratch_buffer_barrier = hal::BufferBarrier { + buffer: &scratch_buffer, + usage: hal::StateTransition { + from: wgpu_types::BufferUses::BOTTOM_LEVEL_ACCELERATION_STRUCTURE_INPUT, + to: wgpu_types::BufferUses::TOP_LEVEL_ACCELERATION_STRUCTURE_INPUT, + }, + }; + cmd_encoder.transition_buffers(iter::once(scratch_buffer_barrier)); + + cmd_encoder.place_acceleration_structure_barrier(hal::AccelerationStructureBarrier { + usage: hal::StateTransition { + from: hal::AccelerationStructureUses::BUILD_OUTPUT, + to: hal::AccelerationStructureUses::BUILD_INPUT, + }, + }); + + cmd_encoder.build_acceleration_structures( + 1, + [hal::BuildAccelerationStructureDescriptor { + mode: hal::AccelerationStructureBuildMode::Build, + flags: tlas_flags, + destination_acceleration_structure: &tlas, + scratch_buffer: &scratch_buffer, + entries: &tlas_entries, + source_acceleration_structure: None, + scratch_buffer_offset: 0, + }], + ); + + cmd_encoder.place_acceleration_structure_barrier(hal::AccelerationStructureBarrier { + usage: hal::StateTransition { + from: hal::AccelerationStructureUses::BUILD_OUTPUT, + to: hal::AccelerationStructureUses::SHADER_INPUT, + }, + }); + + let texture_barrier = hal::TextureBarrier { + texture: &texture, + range: wgpu_types::ImageSubresourceRange::default(), + usage: hal::StateTransition { + from: wgpu_types::TextureUses::UNINITIALIZED, + to: wgpu_types::TextureUses::STORAGE_READ_WRITE, + }, + }; + + cmd_encoder.transition_textures(iter::once(texture_barrier)); + } + + let init_fence_value = 1; + let fence = unsafe { + let mut fence = device.create_fence().unwrap(); + let init_cmd = cmd_encoder.end_encoding().unwrap(); + queue + .submit(&[&init_cmd], &[], (&mut fence, init_fence_value)) + .unwrap(); + device.wait(&fence, init_fence_value, None).unwrap(); + cmd_encoder.reset_all(iter::once(init_cmd)); + fence + }; + + Ok(Self { + instance, + adapter, + surface, + surface_format: surface_config.format, + device, + queue, + pipeline, + contexts: vec![ExecutionContext { + encoder: cmd_encoder, + fence, + fence_value: init_fence_value + 1, + used_views: Vec::new(), + used_cmd_bufs: Vec::new(), + frames_recorded: 0, + }], + context_index: 0, + extent: [window_size.0, window_size.1], + start: Instant::now(), + pipeline_layout, + bind_group, + texture, + instances, + instances_buffer, + blas, + tlas, + scratch_buffer, + time: 0.0, + indices_buffer: indices_buffer.map(|(buf, _)| buf), + vertices_buffer, + uniform_buffer, + texture_view, + bgl, + shader_module, + }) + } + + fn update(&mut self, _event: winit::event::WindowEvent) {} + + fn render(&mut self) { + let ctx = &mut self.contexts[self.context_index]; + + let surface_tex = unsafe { + self.surface + .acquire_texture(None, &ctx.fence) + .unwrap() + .texture + }; + + let target_barrier0 = hal::TextureBarrier { + texture: surface_tex.borrow(), + range: wgpu_types::ImageSubresourceRange::default(), + usage: hal::StateTransition { + from: wgpu_types::TextureUses::UNINITIALIZED, + to: wgpu_types::TextureUses::COPY_DST, + }, + }; + + let instances_buffer_size = + self.instances.len() * size_of::(); + + let tlas_flags = hal::AccelerationStructureBuildFlags::PREFER_FAST_TRACE + | hal::AccelerationStructureBuildFlags::ALLOW_UPDATE; + + self.time += 1.0 / 60.0; + + self.instances[0].set_transform(&Affine3A::from_rotation_y(self.time)); + + unsafe { + let mapping = self + .device + .map_buffer(&self.instances_buffer, 0..instances_buffer_size as u64) + .unwrap(); + ptr::copy_nonoverlapping( + self.instances.as_ptr() as *const u8, + mapping.ptr.as_ptr(), + instances_buffer_size, + ); + self.device.unmap_buffer(&self.instances_buffer); + assert!(mapping.is_coherent); + } + + unsafe { + ctx.encoder.begin_encoding(Some("frame")).unwrap(); + + let instances = hal::AccelerationStructureInstances { + buffer: Some(&self.instances_buffer), + count: self.instances.len() as u32, + offset: 0, + }; + + ctx.encoder + .place_acceleration_structure_barrier(hal::AccelerationStructureBarrier { + usage: hal::StateTransition { + from: hal::AccelerationStructureUses::SHADER_INPUT, + to: hal::AccelerationStructureUses::BUILD_INPUT, + }, + }); + + ctx.encoder.build_acceleration_structures( + 1, + [hal::BuildAccelerationStructureDescriptor { + mode: hal::AccelerationStructureBuildMode::Update, + flags: tlas_flags, + destination_acceleration_structure: &self.tlas, + scratch_buffer: &self.scratch_buffer, + entries: &hal::AccelerationStructureEntries::Instances(instances), + source_acceleration_structure: Some(&self.tlas), + scratch_buffer_offset: 0, + }], + ); + + ctx.encoder + .place_acceleration_structure_barrier(hal::AccelerationStructureBarrier { + usage: hal::StateTransition { + from: hal::AccelerationStructureUses::BUILD_OUTPUT, + to: hal::AccelerationStructureUses::SHADER_INPUT, + }, + }); + + let scratch_buffer_barrier = hal::BufferBarrier { + buffer: &self.scratch_buffer, + usage: hal::StateTransition { + from: wgpu_types::BufferUses::BOTTOM_LEVEL_ACCELERATION_STRUCTURE_INPUT, + to: wgpu_types::BufferUses::TOP_LEVEL_ACCELERATION_STRUCTURE_INPUT, + }, + }; + ctx.encoder + .transition_buffers(iter::once(scratch_buffer_barrier)); + + ctx.encoder.transition_textures(iter::once(target_barrier0)); + } + + let surface_view_desc = hal::TextureViewDescriptor { + label: None, + format: self.surface_format, + dimension: wgpu_types::TextureViewDimension::D2, + usage: wgpu_types::TextureUses::COPY_DST, + range: wgpu_types::ImageSubresourceRange::default(), + }; + let surface_tex_view = unsafe { + self.device + .create_texture_view(surface_tex.borrow(), &surface_view_desc) + .unwrap() + }; + unsafe { + ctx.encoder.begin_compute_pass(&hal::ComputePassDescriptor { + label: None, + timestamp_writes: None, + }); + ctx.encoder.set_compute_pipeline(&self.pipeline); + ctx.encoder + .set_bind_group(&self.pipeline_layout, 0, &self.bind_group, &[]); + ctx.encoder.dispatch([512 / 8, 512 / 8, 1]); + } + + ctx.frames_recorded += 1; + + let target_barrier1 = hal::TextureBarrier { + texture: surface_tex.borrow(), + range: wgpu_types::ImageSubresourceRange::default(), + usage: hal::StateTransition { + from: wgpu_types::TextureUses::COPY_DST, + to: wgpu_types::TextureUses::PRESENT, + }, + }; + let target_barrier2 = hal::TextureBarrier { + texture: &self.texture, + range: wgpu_types::ImageSubresourceRange::default(), + usage: hal::StateTransition { + from: wgpu_types::TextureUses::STORAGE_READ_WRITE, + to: wgpu_types::TextureUses::COPY_SRC, + }, + }; + let target_barrier3 = hal::TextureBarrier { + texture: &self.texture, + range: wgpu_types::ImageSubresourceRange::default(), + usage: hal::StateTransition { + from: wgpu_types::TextureUses::COPY_SRC, + to: wgpu_types::TextureUses::STORAGE_READ_WRITE, + }, + }; + unsafe { + ctx.encoder.end_compute_pass(); + ctx.encoder.transition_textures(iter::once(target_barrier2)); + ctx.encoder.copy_texture_to_texture( + &self.texture, + wgpu_types::TextureUses::COPY_SRC, + surface_tex.borrow(), + std::iter::once(hal::TextureCopy { + src_base: hal::TextureCopyBase { + mip_level: 0, + array_layer: 0, + origin: wgpu_types::Origin3d::ZERO, + aspect: hal::FormatAspects::COLOR, + }, + dst_base: hal::TextureCopyBase { + mip_level: 0, + array_layer: 0, + origin: wgpu_types::Origin3d::ZERO, + aspect: hal::FormatAspects::COLOR, + }, + size: hal::CopyExtent { + width: 512, + height: 512, + depth: 1, + }, + }), + ); + ctx.encoder.transition_textures(iter::once(target_barrier1)); + ctx.encoder.transition_textures(iter::once(target_barrier3)); + } + + unsafe { + let cmd_buf = ctx.encoder.end_encoding().unwrap(); + self.queue + .submit( + &[&cmd_buf], + &[&surface_tex], + (&mut ctx.fence, ctx.fence_value), + ) + .unwrap(); + self.queue.present(&self.surface, surface_tex).unwrap(); + ctx.used_cmd_bufs.push(cmd_buf); + ctx.used_views.push(surface_tex_view); + }; + + log::info!("Context switch from {}", self.context_index); + let old_fence_value = ctx.fence_value; + if self.contexts.len() == 1 { + let hal_desc = hal::CommandEncoderDescriptor { + label: None, + queue: &self.queue, + }; + self.contexts.push(unsafe { + ExecutionContext { + encoder: self.device.create_command_encoder(&hal_desc).unwrap(), + fence: self.device.create_fence().unwrap(), + fence_value: 0, + used_views: Vec::new(), + used_cmd_bufs: Vec::new(), + frames_recorded: 0, + } + }); + } + self.context_index = (self.context_index + 1) % self.contexts.len(); + let next = &mut self.contexts[self.context_index]; + unsafe { + next.wait_and_clear(&self.device); + } + next.fence_value = old_fence_value + 1; + } + + fn exit(mut self) { + unsafe { + { + let ctx = &mut self.contexts[self.context_index]; + self.queue + .submit(&[], &[], (&mut ctx.fence, ctx.fence_value)) + .unwrap(); + } + + for mut ctx in self.contexts { + ctx.wait_and_clear(&self.device); + drop(ctx.encoder); + self.device.destroy_fence(ctx.fence); + } + + self.device.destroy_bind_group(self.bind_group); + self.device.destroy_buffer(self.scratch_buffer); + self.device.destroy_buffer(self.instances_buffer); + if let Some(buffer) = self.indices_buffer { + self.device.destroy_buffer(buffer); + } + self.device.destroy_buffer(self.vertices_buffer); + self.device.destroy_buffer(self.uniform_buffer); + self.device.destroy_acceleration_structure(self.tlas); + self.device.destroy_acceleration_structure(self.blas); + self.device.destroy_texture_view(self.texture_view); + self.device.destroy_texture(self.texture); + self.device.destroy_compute_pipeline(self.pipeline); + self.device.destroy_pipeline_layout(self.pipeline_layout); + self.device.destroy_bind_group_layout(self.bgl); + self.device.destroy_shader_module(self.shader_module); + + self.surface.unconfigure(&self.device); + drop(self.queue); + drop(self.device); + drop(self.surface); + drop(self.adapter); + } + } +} + +cfg_if::cfg_if! { + // Apple + Metal + if #[cfg(all(target_vendor = "apple", feature = "metal"))] { + type Api = hal::api::Metal; + } + // Wasm + Vulkan + else if #[cfg(all(not(target_arch = "wasm32"), feature = "vulkan"))] { + type Api = hal::api::Vulkan; + } + // Windows + DX12 + else if #[cfg(all(windows, feature = "dx12"))] { + type Api = hal::api::Dx12; + } + // Anything + GLES + else if #[cfg(feature = "gles")] { + type Api = hal::api::Gles; + } + // Fallback + else { + type Api = hal::api::Noop; + } +} + +struct App { + example: Option>, + window: Option, +} + +impl ApplicationHandler for App { + fn resumed(&mut self, event_loop: &ActiveEventLoop) { + if self.window.is_some() { + return; + } + let window = event_loop + .create_window( + Window::default_attributes() + .with_title("hal-ray-traced-triangle") + .with_inner_size(winit::dpi::PhysicalSize { + width: 512, + height: 512, + }) + .with_resizable(false) + .with_enabled_buttons(WindowButtons::CLOSE), + ) + .unwrap(); + let example = Example::::init(&window).expect("Selected backend is not supported"); + self.window = Some(window); + self.example = Some(example); + } + + fn exiting(&mut self, _event_loop: &ActiveEventLoop) { + self.example.take().unwrap().exit(); + } + + fn about_to_wait(&mut self, _event_loop: &ActiveEventLoop) { + if let Some(window) = &self.window { + window.request_redraw(); + } + } + + fn window_event( + &mut self, + event_loop: &ActiveEventLoop, + _window_id: winit::window::WindowId, + event: winit::event::WindowEvent, + ) { + event_loop.set_control_flow(ControlFlow::Poll); + + match event { + winit::event::WindowEvent::CloseRequested => { + event_loop.exit(); + } + winit::event::WindowEvent::KeyboardInput { event, .. } + if event.physical_key + == winit::keyboard::PhysicalKey::Code(winit::keyboard::KeyCode::Escape) => + { + event_loop.exit(); + } + winit::event::WindowEvent::RedrawRequested => { + let ex = self.example.as_mut().unwrap(); + ex.render(); + } + _ => { + self.example.as_mut().unwrap().update(event); + } + } + } +} + +fn main() { + env_logger::init(); + + let event_loop = winit::event_loop::EventLoop::new().unwrap(); + let mut app = App { + example: None, + window: None, + }; + event_loop.run_app(&mut app).unwrap(); +} diff --git a/third_party/wgpu-hal-29.0.4/examples/ray-traced-triangle/shader.wgsl b/third_party/wgpu-hal-29.0.4/examples/ray-traced-triangle/shader.wgsl new file mode 100644 index 0000000..9710ca6 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/examples/ray-traced-triangle/shader.wgsl @@ -0,0 +1,39 @@ +enable wgpu_ray_query; + +struct Uniforms { + view_inv: mat4x4, + proj_inv: mat4x4, +}; +@group(0) @binding(0) +var uniforms: Uniforms; + +@group(0) @binding(1) +var output: texture_storage_2d; + +@group(0) @binding(2) +var acc_struct: acceleration_structure; + +@compute @workgroup_size(8, 8) +fn main(@builtin(global_invocation_id) global_id: vec3) { + let target_size = textureDimensions(output); + + let pixel_center = vec2(global_id.xy) + vec2(0.5); + let in_uv = pixel_center / vec2(target_size.xy); + let d = in_uv * 2.0 - 1.0; + + let origin = (uniforms.view_inv * vec4(0.0, 0.0, 0.0, 1.0)).xyz; + let temp = uniforms.proj_inv * vec4(d.x, d.y, 1.0, 1.0); + let direction = (uniforms.view_inv * vec4(normalize(temp.xyz), 0.0)).xyz; + + var rq: ray_query; + rayQueryInitialize(&rq, acc_struct, RayDesc(0u, 0xFFu, 0.1, 200.0, origin, direction)); + rayQueryProceed(&rq); + + var color = vec4(0.0, 0.0, 0.0, 1.0); + let intersection = rayQueryGetCommittedIntersection(&rq); + if intersection.kind != RAY_QUERY_INTERSECTION_NONE { + color = vec4(intersection.barycentrics, 1.0 - intersection.barycentrics.x - intersection.barycentrics.y, 1.0); + } + + textureStore(output, global_id.xy, color); +} \ No newline at end of file diff --git a/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/conv.rs b/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/conv.rs new file mode 100644 index 0000000..6e27081 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/conv.rs @@ -0,0 +1,295 @@ +use alloc::string::String; +use std::{ffi::OsString, os::windows::ffi::OsStringExt}; + +use windows::Win32::Graphics::Dxgi; + +// Helper to convert DXGI adapter name to a normal string +pub fn map_adapter_name(name: [u16; 128]) -> String { + let len = name.iter().take_while(|&&c| c != 0).count(); + let name = OsString::from_wide(&name[..len]); + name.to_string_lossy().into_owned() +} + +pub fn map_texture_format_failable( + format: wgt::TextureFormat, +) -> Option { + use wgt::TextureFormat as Tf; + use Dxgi::Common::*; + + Some(match format { + Tf::R8Unorm => DXGI_FORMAT_R8_UNORM, + Tf::R8Snorm => DXGI_FORMAT_R8_SNORM, + Tf::R8Uint => DXGI_FORMAT_R8_UINT, + Tf::R8Sint => DXGI_FORMAT_R8_SINT, + Tf::R16Uint => DXGI_FORMAT_R16_UINT, + Tf::R16Sint => DXGI_FORMAT_R16_SINT, + Tf::R16Unorm => DXGI_FORMAT_R16_UNORM, + Tf::R16Snorm => DXGI_FORMAT_R16_SNORM, + Tf::R16Float => DXGI_FORMAT_R16_FLOAT, + Tf::Rg8Unorm => DXGI_FORMAT_R8G8_UNORM, + Tf::Rg8Snorm => DXGI_FORMAT_R8G8_SNORM, + Tf::Rg8Uint => DXGI_FORMAT_R8G8_UINT, + Tf::Rg8Sint => DXGI_FORMAT_R8G8_SINT, + Tf::Rg16Unorm => DXGI_FORMAT_R16G16_UNORM, + Tf::Rg16Snorm => DXGI_FORMAT_R16G16_SNORM, + Tf::R32Uint => DXGI_FORMAT_R32_UINT, + Tf::R32Sint => DXGI_FORMAT_R32_SINT, + Tf::R32Float => DXGI_FORMAT_R32_FLOAT, + Tf::Rg16Uint => DXGI_FORMAT_R16G16_UINT, + Tf::Rg16Sint => DXGI_FORMAT_R16G16_SINT, + Tf::Rg16Float => DXGI_FORMAT_R16G16_FLOAT, + Tf::Rgba8Unorm => DXGI_FORMAT_R8G8B8A8_UNORM, + Tf::Rgba8UnormSrgb => DXGI_FORMAT_R8G8B8A8_UNORM_SRGB, + Tf::Bgra8UnormSrgb => DXGI_FORMAT_B8G8R8A8_UNORM_SRGB, + Tf::Rgba8Snorm => DXGI_FORMAT_R8G8B8A8_SNORM, + Tf::Bgra8Unorm => DXGI_FORMAT_B8G8R8A8_UNORM, + Tf::Rgba8Uint => DXGI_FORMAT_R8G8B8A8_UINT, + Tf::Rgba8Sint => DXGI_FORMAT_R8G8B8A8_SINT, + Tf::Rgb9e5Ufloat => DXGI_FORMAT_R9G9B9E5_SHAREDEXP, + Tf::Rgb10a2Uint => DXGI_FORMAT_R10G10B10A2_UINT, + Tf::Rgb10a2Unorm => DXGI_FORMAT_R10G10B10A2_UNORM, + Tf::Rg11b10Ufloat => DXGI_FORMAT_R11G11B10_FLOAT, + Tf::R64Uint => DXGI_FORMAT_R32G32_UINT, // R64 emulated by R32G32 + Tf::Rg32Uint => DXGI_FORMAT_R32G32_UINT, + Tf::Rg32Sint => DXGI_FORMAT_R32G32_SINT, + Tf::Rg32Float => DXGI_FORMAT_R32G32_FLOAT, + Tf::Rgba16Uint => DXGI_FORMAT_R16G16B16A16_UINT, + Tf::Rgba16Sint => DXGI_FORMAT_R16G16B16A16_SINT, + Tf::Rgba16Unorm => DXGI_FORMAT_R16G16B16A16_UNORM, + Tf::Rgba16Snorm => DXGI_FORMAT_R16G16B16A16_SNORM, + Tf::Rgba16Float => DXGI_FORMAT_R16G16B16A16_FLOAT, + Tf::Rgba32Uint => DXGI_FORMAT_R32G32B32A32_UINT, + Tf::Rgba32Sint => DXGI_FORMAT_R32G32B32A32_SINT, + Tf::Rgba32Float => DXGI_FORMAT_R32G32B32A32_FLOAT, + Tf::Stencil8 => DXGI_FORMAT_D24_UNORM_S8_UINT, + Tf::Depth16Unorm => DXGI_FORMAT_D16_UNORM, + Tf::Depth24Plus => DXGI_FORMAT_D24_UNORM_S8_UINT, + Tf::Depth24PlusStencil8 => DXGI_FORMAT_D24_UNORM_S8_UINT, + Tf::Depth32Float => DXGI_FORMAT_D32_FLOAT, + Tf::Depth32FloatStencil8 => DXGI_FORMAT_D32_FLOAT_S8X24_UINT, + Tf::NV12 => DXGI_FORMAT_NV12, + Tf::P010 => DXGI_FORMAT_P010, + Tf::Bc1RgbaUnorm => DXGI_FORMAT_BC1_UNORM, + Tf::Bc1RgbaUnormSrgb => DXGI_FORMAT_BC1_UNORM_SRGB, + Tf::Bc2RgbaUnorm => DXGI_FORMAT_BC2_UNORM, + Tf::Bc2RgbaUnormSrgb => DXGI_FORMAT_BC2_UNORM_SRGB, + Tf::Bc3RgbaUnorm => DXGI_FORMAT_BC3_UNORM, + Tf::Bc3RgbaUnormSrgb => DXGI_FORMAT_BC3_UNORM_SRGB, + Tf::Bc4RUnorm => DXGI_FORMAT_BC4_UNORM, + Tf::Bc4RSnorm => DXGI_FORMAT_BC4_SNORM, + Tf::Bc5RgUnorm => DXGI_FORMAT_BC5_UNORM, + Tf::Bc5RgSnorm => DXGI_FORMAT_BC5_SNORM, + Tf::Bc6hRgbUfloat => DXGI_FORMAT_BC6H_UF16, + Tf::Bc6hRgbFloat => DXGI_FORMAT_BC6H_SF16, + Tf::Bc7RgbaUnorm => DXGI_FORMAT_BC7_UNORM, + Tf::Bc7RgbaUnormSrgb => DXGI_FORMAT_BC7_UNORM_SRGB, + Tf::Etc2Rgb8Unorm + | Tf::Etc2Rgb8UnormSrgb + | Tf::Etc2Rgb8A1Unorm + | Tf::Etc2Rgb8A1UnormSrgb + | Tf::Etc2Rgba8Unorm + | Tf::Etc2Rgba8UnormSrgb + | Tf::EacR11Unorm + | Tf::EacR11Snorm + | Tf::EacRg11Unorm + | Tf::EacRg11Snorm + | Tf::Astc { + block: _, + channel: _, + } => return None, + }) +} + +pub fn map_texture_format(format: wgt::TextureFormat) -> Dxgi::Common::DXGI_FORMAT { + match map_texture_format_failable(format) { + Some(f) => f, + None => unreachable!(), + } +} + +// Note: DXGI doesn't allow sRGB format on the swapchain, +// but creating RTV of swapchain buffers with sRGB works. +pub fn map_texture_format_nosrgb(format: wgt::TextureFormat) -> Dxgi::Common::DXGI_FORMAT { + match format { + wgt::TextureFormat::Bgra8UnormSrgb => Dxgi::Common::DXGI_FORMAT_B8G8R8A8_UNORM, + wgt::TextureFormat::Rgba8UnormSrgb => Dxgi::Common::DXGI_FORMAT_R8G8B8A8_UNORM, + _ => map_texture_format(format), + } +} + +// SRV and UAV can't use the depth or typeless formats +// see https://microsoft.github.io/DirectX-Specs/d3d/PlanarDepthStencilDDISpec.html#view-creation +pub fn map_texture_format_for_srv_uav( + format: wgt::TextureFormat, + aspect: crate::FormatAspects, +) -> Option { + Some(match (format, aspect) { + (wgt::TextureFormat::Depth16Unorm, crate::FormatAspects::DEPTH) => { + Dxgi::Common::DXGI_FORMAT_R16_UNORM + } + (wgt::TextureFormat::Depth32Float, crate::FormatAspects::DEPTH) => { + Dxgi::Common::DXGI_FORMAT_R32_FLOAT + } + (wgt::TextureFormat::Depth32FloatStencil8, crate::FormatAspects::DEPTH) => { + Dxgi::Common::DXGI_FORMAT_R32_FLOAT_X8X24_TYPELESS + } + ( + wgt::TextureFormat::Depth24Plus | wgt::TextureFormat::Depth24PlusStencil8, + crate::FormatAspects::DEPTH, + ) => Dxgi::Common::DXGI_FORMAT_R24_UNORM_X8_TYPELESS, + + (wgt::TextureFormat::Depth32FloatStencil8, crate::FormatAspects::STENCIL) => { + Dxgi::Common::DXGI_FORMAT_X32_TYPELESS_G8X24_UINT + } + ( + wgt::TextureFormat::Stencil8 | wgt::TextureFormat::Depth24PlusStencil8, + crate::FormatAspects::STENCIL, + ) => Dxgi::Common::DXGI_FORMAT_X24_TYPELESS_G8_UINT, + + (_, crate::FormatAspects::DEPTH) + | (_, crate::FormatAspects::STENCIL) + | (_, crate::FormatAspects::DEPTH_STENCIL) => return None, + + _ => map_texture_format(format), + }) +} + +// see https://microsoft.github.io/DirectX-Specs/d3d/PlanarDepthStencilDDISpec.html#planar-layout-for-staging-from-buffer +pub fn map_texture_format_for_copy( + format: wgt::TextureFormat, + aspect: crate::FormatAspects, +) -> Option { + Some(match (format, aspect) { + (wgt::TextureFormat::Depth16Unorm, crate::FormatAspects::DEPTH) => { + Dxgi::Common::DXGI_FORMAT_R16_UNORM + } + ( + wgt::TextureFormat::Depth32Float | wgt::TextureFormat::Depth32FloatStencil8, + crate::FormatAspects::DEPTH, + ) => Dxgi::Common::DXGI_FORMAT_R32_FLOAT, + + ( + wgt::TextureFormat::Stencil8 + | wgt::TextureFormat::Depth24PlusStencil8 + | wgt::TextureFormat::Depth32FloatStencil8, + crate::FormatAspects::STENCIL, + ) => Dxgi::Common::DXGI_FORMAT_R8_UINT, + + (format, crate::FormatAspects::COLOR) => map_texture_format(format), + + _ => return None, + }) +} + +pub fn map_texture_format_for_resource( + format: wgt::TextureFormat, + usage: wgt::TextureUses, + has_view_formats: bool, + casting_fully_typed_format_supported: bool, +) -> Dxgi::Common::DXGI_FORMAT { + use wgt::TextureFormat as Tf; + use Dxgi::Common::*; + + if casting_fully_typed_format_supported { + map_texture_format(format) + + // We might view this resource as srgb or non-srgb + } else if has_view_formats { + match format { + Tf::Rgba8Unorm | Tf::Rgba8UnormSrgb => DXGI_FORMAT_R8G8B8A8_TYPELESS, + Tf::Bgra8Unorm | Tf::Bgra8UnormSrgb => DXGI_FORMAT_B8G8R8A8_TYPELESS, + Tf::Bc1RgbaUnorm | Tf::Bc1RgbaUnormSrgb => DXGI_FORMAT_BC1_TYPELESS, + Tf::Bc2RgbaUnorm | Tf::Bc2RgbaUnormSrgb => DXGI_FORMAT_BC2_TYPELESS, + Tf::Bc3RgbaUnorm | Tf::Bc3RgbaUnormSrgb => DXGI_FORMAT_BC3_TYPELESS, + Tf::Bc7RgbaUnorm | Tf::Bc7RgbaUnormSrgb => DXGI_FORMAT_BC7_TYPELESS, + format => map_texture_format(format), + } + + // We might view this resource as SRV/UAV but also as DSV + } else if format.is_depth_stencil_format() + && usage.intersects( + wgt::TextureUses::RESOURCE + | wgt::TextureUses::STORAGE_READ_ONLY + | wgt::TextureUses::STORAGE_WRITE_ONLY + | wgt::TextureUses::STORAGE_READ_WRITE, + ) + { + match format { + Tf::Depth16Unorm => DXGI_FORMAT_R16_TYPELESS, + Tf::Depth32Float => DXGI_FORMAT_R32_TYPELESS, + Tf::Depth32FloatStencil8 => DXGI_FORMAT_R32G8X24_TYPELESS, + Tf::Stencil8 | Tf::Depth24Plus | Tf::Depth24PlusStencil8 => DXGI_FORMAT_R24G8_TYPELESS, + _ => unreachable!(), + } + } else { + map_texture_format(format) + } +} + +pub fn map_index_format(format: wgt::IndexFormat) -> Dxgi::Common::DXGI_FORMAT { + match format { + wgt::IndexFormat::Uint16 => Dxgi::Common::DXGI_FORMAT_R16_UINT, + wgt::IndexFormat::Uint32 => Dxgi::Common::DXGI_FORMAT_R32_UINT, + } +} + +pub fn map_vertex_format(format: wgt::VertexFormat) -> Dxgi::Common::DXGI_FORMAT { + use wgt::VertexFormat as Vf; + use Dxgi::Common::*; + + match format { + Vf::Unorm8 => DXGI_FORMAT_R8_UNORM, + Vf::Snorm8 => DXGI_FORMAT_R8_SNORM, + Vf::Uint8 => DXGI_FORMAT_R8_UINT, + Vf::Sint8 => DXGI_FORMAT_R8_SINT, + Vf::Unorm8x2 => DXGI_FORMAT_R8G8_UNORM, + Vf::Snorm8x2 => DXGI_FORMAT_R8G8_SNORM, + Vf::Uint8x2 => DXGI_FORMAT_R8G8_UINT, + Vf::Sint8x2 => DXGI_FORMAT_R8G8_SINT, + Vf::Unorm8x4 => DXGI_FORMAT_R8G8B8A8_UNORM, + Vf::Snorm8x4 => DXGI_FORMAT_R8G8B8A8_SNORM, + Vf::Uint8x4 => DXGI_FORMAT_R8G8B8A8_UINT, + Vf::Sint8x4 => DXGI_FORMAT_R8G8B8A8_SINT, + Vf::Unorm16 => DXGI_FORMAT_R16_UNORM, + Vf::Snorm16 => DXGI_FORMAT_R16_SNORM, + Vf::Uint16 => DXGI_FORMAT_R16_UINT, + Vf::Sint16 => DXGI_FORMAT_R16_SINT, + Vf::Float16 => DXGI_FORMAT_R16_FLOAT, + Vf::Unorm16x2 => DXGI_FORMAT_R16G16_UNORM, + Vf::Snorm16x2 => DXGI_FORMAT_R16G16_SNORM, + Vf::Uint16x2 => DXGI_FORMAT_R16G16_UINT, + Vf::Sint16x2 => DXGI_FORMAT_R16G16_SINT, + Vf::Float16x2 => DXGI_FORMAT_R16G16_FLOAT, + Vf::Unorm16x4 => DXGI_FORMAT_R16G16B16A16_UNORM, + Vf::Snorm16x4 => DXGI_FORMAT_R16G16B16A16_SNORM, + Vf::Uint16x4 => DXGI_FORMAT_R16G16B16A16_UINT, + Vf::Sint16x4 => DXGI_FORMAT_R16G16B16A16_SINT, + Vf::Float16x4 => DXGI_FORMAT_R16G16B16A16_FLOAT, + Vf::Uint32 => DXGI_FORMAT_R32_UINT, + Vf::Sint32 => DXGI_FORMAT_R32_SINT, + Vf::Float32 => DXGI_FORMAT_R32_FLOAT, + Vf::Uint32x2 => DXGI_FORMAT_R32G32_UINT, + Vf::Sint32x2 => DXGI_FORMAT_R32G32_SINT, + Vf::Float32x2 => DXGI_FORMAT_R32G32_FLOAT, + Vf::Uint32x3 => DXGI_FORMAT_R32G32B32_UINT, + Vf::Sint32x3 => DXGI_FORMAT_R32G32B32_SINT, + Vf::Float32x3 => DXGI_FORMAT_R32G32B32_FLOAT, + Vf::Uint32x4 => DXGI_FORMAT_R32G32B32A32_UINT, + Vf::Sint32x4 => DXGI_FORMAT_R32G32B32A32_SINT, + Vf::Float32x4 => DXGI_FORMAT_R32G32B32A32_FLOAT, + Vf::Unorm10_10_10_2 => DXGI_FORMAT_R10G10B10A2_UNORM, + Vf::Unorm8x4Bgra => DXGI_FORMAT_B8G8R8A8_UNORM, + Vf::Float64 | Vf::Float64x2 | Vf::Float64x3 | Vf::Float64x4 => unimplemented!(), + } +} + +pub fn map_acomposite_alpha_mode(mode: wgt::CompositeAlphaMode) -> Dxgi::Common::DXGI_ALPHA_MODE { + match mode { + wgt::CompositeAlphaMode::PreMultiplied => Dxgi::Common::DXGI_ALPHA_MODE_PREMULTIPLIED, + wgt::CompositeAlphaMode::PostMultiplied => Dxgi::Common::DXGI_ALPHA_MODE_STRAIGHT, + wgt::CompositeAlphaMode::Opaque => Dxgi::Common::DXGI_ALPHA_MODE_IGNORE, + wgt::CompositeAlphaMode::Auto | wgt::CompositeAlphaMode::Inherit => { + Dxgi::Common::DXGI_ALPHA_MODE_UNSPECIFIED + } + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/exception.rs b/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/exception.rs new file mode 100644 index 0000000..ac7d55d --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/exception.rs @@ -0,0 +1,97 @@ +use alloc::{borrow::Cow, string::String}; + +use parking_lot::Mutex; +use windows::Win32::{Foundation, System::Diagnostics::Debug}; + +// This is a mutex as opposed to an atomic as we need to completely +// lock everyone out until we have registered or unregistered the +// exception handler, otherwise really nasty races could happen. +// +// By routing all the registration through these functions we can guarantee +// there is either 1 or 0 exception handlers registered, not multiple. +static EXCEPTION_HANDLER_COUNT: Mutex = Mutex::new(0); + +pub fn register_exception_handler() { + let mut count_guard = EXCEPTION_HANDLER_COUNT.lock(); + if *count_guard == 0 { + unsafe { Debug::AddVectoredExceptionHandler(0, Some(output_debug_string_handler)) }; + } + *count_guard += 1; +} + +pub fn unregister_exception_handler() { + let mut count_guard = EXCEPTION_HANDLER_COUNT.lock(); + if *count_guard == 1 { + unsafe { Debug::RemoveVectoredExceptionHandler(output_debug_string_handler as *mut _) }; + } + *count_guard -= 1; +} + +const MESSAGE_PREFIXES: &[(&str, log::Level)] = &[ + ("CORRUPTION", log::Level::Error), + ("ERROR", log::Level::Error), + ("WARNING", log::Level::Warn), + // We intentionally suppress "INFO" messages down to debug + // so that users are not innundated with info messages from the runtime. + ("INFO", log::Level::Debug), + ("MESSAGE", log::Level::Trace), +]; + +unsafe extern "system" fn output_debug_string_handler( + exception_info: *mut Debug::EXCEPTION_POINTERS, +) -> i32 { + // See https://stackoverflow.com/a/41480827 + let record = unsafe { &*(*exception_info).ExceptionRecord }; + if record.NumberParameters != 2 { + return Debug::EXCEPTION_CONTINUE_SEARCH; + } + let message = match record.ExceptionCode { + Foundation::DBG_PRINTEXCEPTION_C => { + String::from_utf8_lossy(bytemuck::cast_slice(&record.ExceptionInformation)) + } + Foundation::DBG_PRINTEXCEPTION_WIDE_C => Cow::Owned(String::from_utf16_lossy( + bytemuck::cast_slice(&record.ExceptionInformation), + )), + _ => return Debug::EXCEPTION_CONTINUE_SEARCH, + }; + + let message = match message.strip_prefix("D3D12 ") { + Some(msg) => msg + .trim_end_matches("\n\0") + .trim_end_matches("[ STATE_CREATION WARNING #0: UNKNOWN]"), + None => return Debug::EXCEPTION_CONTINUE_SEARCH, + }; + + let (message, level) = match MESSAGE_PREFIXES + .iter() + .find(|&&(prefix, _)| message.starts_with(prefix)) + { + Some(&(prefix, level)) => (&message[prefix.len() + 2..], level), + None => (message, log::Level::Debug), + }; + + if level == log::Level::Warn && message.contains("#82") { + // This is are useless spammy warnings (#820, #821): + // "The application did not pass any clear value to resource creation" + return Debug::EXCEPTION_CONTINUE_SEARCH; + } + + if level == log::Level::Warn && message.contains("DRAW_EMPTY_SCISSOR_RECTANGLE") { + // This is normal, WebGPU allows passing empty scissor rectangles. + return Debug::EXCEPTION_CONTINUE_SEARCH; + } + + let _ = std::panic::catch_unwind(|| { + log::log!(level, "{message}"); + }); + + #[cfg(feature = "validation_canary")] + if cfg!(debug_assertions) && level == log::Level::Error { + use alloc::string::ToString as _; + + // Set canary and continue + crate::VALIDATION_CANARY.add(message.to_string()); + } + + Debug::EXCEPTION_CONTINUE_EXECUTION +} diff --git a/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/factory.rs b/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/factory.rs new file mode 100644 index 0000000..ac050bc --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/factory.rs @@ -0,0 +1,181 @@ +use alloc::{string::String, vec::Vec}; +use core::ops::Deref; + +use windows::{core::Interface as _, Win32::Graphics::Dxgi}; + +use crate::dx12::DxgiLib; + +use super::result::HResult as _; + +// We can rely on the presence of DXGI 1.4 since D3D12 requires WDDM 2.0, Windows 10 (1507), and so does DXGI 1.4. + +fn should_keep_adapter(adapter: &Dxgi::IDXGIAdapter1) -> bool { + let desc = unsafe { adapter.GetDesc1() }.unwrap(); + + // The Intel Haswell family of iGPUs had support for the D3D12 API but it was later + // removed due to a security vulnerability. + // + // We are explicitly filtering out all the devices in the family because we are now + // getting reports of device loss at a later time than at device creation time (`D3D12CreateDevice`). + // + // See https://www.intel.com/content/www/us/en/support/articles/000057520/graphics.html + // This list of device IDs is from https://dgpu-docs.intel.com/devices/hardware-table.html + let haswell_device_ids = [ + 0x0422, 0x0426, 0x042A, 0x042B, 0x042E, 0x0C22, 0x0C26, 0x0C2A, 0x0C2B, 0x0C2E, 0x0A22, + 0x0A2A, 0x0A2B, 0x0D2A, 0x0D2B, 0x0D2E, 0x0A26, 0x0A2E, 0x0D22, 0x0D26, 0x0412, 0x0416, + 0x0D12, 0x041A, 0x041B, 0x0C12, 0x0C16, 0x0C1A, 0x0C1B, 0x0C1E, 0x0A12, 0x0A1A, 0x0A1B, + 0x0D16, 0x0D1A, 0x0D1B, 0x0D1E, 0x041E, 0x0A16, 0x0A1E, 0x0402, 0x0406, 0x040A, 0x040B, + 0x040E, 0x0C02, 0x0C06, 0x0C0A, 0x0C0B, 0x0C0E, 0x0A02, 0x0A06, 0x0A0A, 0x0A0B, 0x0A0E, + 0x0D02, 0x0D06, 0x0D0A, 0x0D0B, 0x0D0E, + ]; + if desc.VendorId == 0x8086 && haswell_device_ids.contains(&desc.DeviceId) { + return false; + } + + // If run completely headless, windows will show two different WARP adapters, one + // which is lying about being an integrated card. This is so that programs + // that ignore software adapters will actually run on headless/gpu-less machines. + // + // We don't want that and discourage that kind of filtering anyway, so we skip the integrated WARP. + if desc.VendorId == 5140 + && !Dxgi::DXGI_ADAPTER_FLAG(desc.Flags as i32).contains(Dxgi::DXGI_ADAPTER_FLAG_SOFTWARE) + { + let adapter_name = super::conv::map_adapter_name(desc.Description); + if adapter_name.contains("Microsoft Basic Render Driver") { + return false; + } + } + + true +} + +#[derive(Clone, Debug)] +pub enum DxgiAdapter { + /// Provided by DXGI 1.4 + Adapter3(Dxgi::IDXGIAdapter3), + /// Provided by DXGI 1.6 + Adapter4(Dxgi::IDXGIAdapter4), +} + +impl DxgiAdapter { + pub fn query_video_memory_info( + &self, + group: Dxgi::DXGI_MEMORY_SEGMENT_GROUP, + ) -> Result { + let mut info = Dxgi::DXGI_QUERY_VIDEO_MEMORY_INFO::default(); + unsafe { self.QueryVideoMemoryInfo(0, group, &mut info) } + .into_device_result("QueryVideoMemoryInfo")?; + Ok(info) + } +} + +impl Deref for DxgiAdapter { + type Target = Dxgi::IDXGIAdapter3; + + fn deref(&self) -> &Self::Target { + match self { + DxgiAdapter::Adapter3(a) => a, + DxgiAdapter::Adapter4(a) => a, + } + } +} + +pub fn enumerate_adapters(factory: DxgiFactory) -> Vec { + let mut adapters = Vec::with_capacity(8); + + for cur_index in 0.. { + profiling::scope!("IDXGIFactory1::EnumAdapters1"); + let adapter1: Dxgi::IDXGIAdapter1 = match unsafe { factory.EnumAdapters1(cur_index) } { + Ok(a) => a, + Err(e) if e.code() == Dxgi::DXGI_ERROR_NOT_FOUND => break, + Err(e) => { + log::error!("Failed enumerating adapters: {e}"); + break; + } + }; + + if !should_keep_adapter(&adapter1) { + continue; + } + + if let Ok(adapter4) = adapter1.cast::() { + adapters.push(DxgiAdapter::Adapter4(adapter4)); + } else { + let adapter3 = adapter1.cast::().unwrap(); + adapters.push(DxgiAdapter::Adapter3(adapter3)); + } + } + + adapters +} + +#[derive(Clone, Debug)] +pub enum DxgiFactory { + /// Provided by DXGI 1.4 + Factory4(Dxgi::IDXGIFactory4), + /// Provided by DXGI 1.5 + Factory5(Dxgi::IDXGIFactory5), + /// Provided by DXGI 1.6 + Factory6(Dxgi::IDXGIFactory6), +} + +impl Deref for DxgiFactory { + type Target = Dxgi::IDXGIFactory4; + + fn deref(&self) -> &Self::Target { + match self { + DxgiFactory::Factory4(f) => f, + DxgiFactory::Factory5(f) => f, + DxgiFactory::Factory6(f) => f, + } + } +} + +impl DxgiFactory { + pub fn as_factory5(&self) -> Option<&Dxgi::IDXGIFactory5> { + match self { + Self::Factory4(_) => None, + Self::Factory5(f) => Some(f), + Self::Factory6(f) => Some(f), + } + } +} + +pub fn create_factory( + instance_flags: wgt::InstanceFlags, +) -> Result<(DxgiLib, DxgiFactory), crate::InstanceError> { + let lib_dxgi = DxgiLib::new().map_err(|e| { + crate::InstanceError::with_source(String::from("failed to load dxgi.dll"), e) + })?; + + let mut factory_flags = Dxgi::DXGI_CREATE_FACTORY_FLAGS::default(); + + if instance_flags.contains(wgt::InstanceFlags::VALIDATION) { + // The `DXGI_CREATE_FACTORY_DEBUG` flag is only allowed to be passed to + // `CreateDXGIFactory2` if the debug interface is actually available. So + // we check for whether it exists first. + if let Ok(Some(_)) = lib_dxgi.debug_interface1() { + factory_flags |= Dxgi::DXGI_CREATE_FACTORY_DEBUG; + } + } + + let factory4 = match lib_dxgi.create_factory4(factory_flags) { + Ok(factory) => factory, + Err(err) => { + return Err(crate::InstanceError::with_source( + String::from("IDXGIFactory4 creation failed"), + err, + )); + } + }; + + if let Ok(factory6) = factory4.cast::() { + return Ok((lib_dxgi, DxgiFactory::Factory6(factory6))); + } + + if let Ok(factory5) = factory4.cast::() { + return Ok((lib_dxgi, DxgiFactory::Factory5(factory5))); + } + + Ok((lib_dxgi, DxgiFactory::Factory4(factory4))) +} diff --git a/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/mod.rs b/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/mod.rs new file mode 100644 index 0000000..5cc6af6 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/mod.rs @@ -0,0 +1,6 @@ +pub mod conv; +pub mod exception; +pub mod factory; +pub mod name; +pub mod result; +pub mod time; diff --git a/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/name.rs b/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/name.rs new file mode 100644 index 0000000..d120e4e --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/name.rs @@ -0,0 +1,24 @@ +use windows::Win32::Graphics::Direct3D12::ID3D12Object; + +use crate::auxil::dxgi::result::HResult; + +/// Helper trait for setting the name of a D3D12 object. +/// +/// This is implemented on all types that can be converted to an [`ID3D12Object`]. +pub trait ObjectExt { + fn set_name(&self, name: &str) -> Result<(), crate::DeviceError>; +} + +impl ObjectExt for T +where + // Windows impls `From` for all parent interfaces, so we can use that to convert to ID3D12Object. + // + // This includes implementations for references. + for<'a> &'a ID3D12Object: From<&'a T>, +{ + fn set_name(&self, name: &str) -> Result<(), crate::DeviceError> { + let name = windows::core::HSTRING::from(name); + let object: &ID3D12Object = self.into(); + unsafe { object.SetName(&name).into_device_result("SetName") } + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/result.rs b/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/result.rs new file mode 100644 index 0000000..f3d77e2 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/result.rs @@ -0,0 +1,28 @@ +use windows::Win32::{Foundation, Graphics::Dxgi}; + +pub(crate) trait HResult { + fn into_device_result(self, description: &str) -> Result; +} +impl HResult for windows::core::Result { + fn into_device_result(self, description: &str) -> Result { + #![allow(unreachable_code)] + + self.map_err(|err| { + log::error!("{description} failed: {err}"); + + match err.code() { + Foundation::E_OUTOFMEMORY => crate::DeviceError::OutOfMemory, + Dxgi::DXGI_ERROR_DEVICE_RESET | Dxgi::DXGI_ERROR_DEVICE_REMOVED => { + #[cfg(feature = "device_lost_panic")] + panic!("{description} failed: Device lost ({err})"); + crate::DeviceError::Lost + } + _ => { + #[cfg(feature = "internal_error_panic")] + panic!("{description} failed: {err}"); + crate::DeviceError::Unexpected + } + } + }) + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/time.rs b/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/time.rs new file mode 100644 index 0000000..6b4c60f --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/auxil/dxgi/time.rs @@ -0,0 +1,97 @@ +#![allow(dead_code)] // IPresentationManager is unused currently + +use windows::Win32::System::Performance::{QueryPerformanceCounter, QueryPerformanceFrequency}; + +pub enum PresentationTimer { + /// DXGI uses [`QueryPerformanceCounter()`] + Dxgi { + /// How many ticks of QPC per second + frequency: u64, + }, + /// [`IPresentationManager`] uses [`QueryInterruptTimePrecise()`] + /// + /// [`IPresentationManager`]: https://microsoft.github.io/windows-docs-rs/doc/windows/Win32/Graphics/CompositionSwapchain/struct.IPresentationManager.html + /// [`QueryInterruptTimePrecise()`]: https://microsoft.github.io/windows-docs-rs/doc/windows/Win32/System/WindowsProgramming/fn.QueryInterruptTimePrecise.html + #[allow(non_snake_case)] + IPresentationManager { + fnQueryInterruptTimePrecise: unsafe extern "system" fn(*mut u64), + }, +} + +impl core::fmt::Debug for PresentationTimer { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match *self { + Self::Dxgi { frequency } => f + .debug_struct("DXGI") + .field("frequency", &frequency) + .finish(), + Self::IPresentationManager { + fnQueryInterruptTimePrecise, + } => f + .debug_struct("IPresentationManager") + .field( + "QueryInterruptTimePrecise", + &(fnQueryInterruptTimePrecise as usize), + ) + .finish(), + } + } +} + +impl PresentationTimer { + /// Create a presentation timer using QueryPerformanceFrequency (what DXGI uses for presentation times) + pub fn new_dxgi() -> Self { + let mut frequency = 0; + unsafe { QueryPerformanceFrequency(&mut frequency) }.unwrap(); + + Self::Dxgi { + frequency: frequency + .try_into() + .expect("Frequency should not be negative"), + } + } + + /// Create a presentation timer using QueryInterruptTimePrecise (what IPresentationManager uses for presentation times) + /// + /// Panics if QueryInterruptTimePrecise isn't found (below Win10) + pub fn new_ipresentation_manager() -> Self { + // We need to load this explicitly, as QueryInterruptTimePrecise is only available on Windows 10+ + // + // Docs say it's in kernel32.dll, but it's actually in kernelbase.dll. + // api-ms-win-core-realtime-l1-1-1.dll + let kernelbase = + libloading::os::windows::Library::open_already_loaded("kernelbase.dll").unwrap(); + // No concerns about lifetimes here as kernelbase is always there. + let ptr = unsafe { + kernelbase + .get(c"QueryInterruptTimePrecise".to_bytes()) + .unwrap() + }; + Self::IPresentationManager { + fnQueryInterruptTimePrecise: *ptr, + } + } + + /// Gets the current time in nanoseconds. + pub fn get_timestamp_ns(&self) -> u128 { + // Always do u128 math _after_ hitting the timing function. + match *self { + PresentationTimer::Dxgi { frequency } => { + let mut counter = 0; + unsafe { QueryPerformanceCounter(&mut counter) }.unwrap(); + + // counter * (1_000_000_000 / freq) but re-ordered to make more precise + (counter as u128 * 1_000_000_000) / frequency as u128 + } + PresentationTimer::IPresentationManager { + fnQueryInterruptTimePrecise, + } => { + let mut counter = 0; + unsafe { fnQueryInterruptTimePrecise(&mut counter) }; + + // QueryInterruptTimePrecise uses units of 100ns for its tick. + counter as u128 * 100 + } + } + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/auxil/mod.rs b/third_party/wgpu-hal-29.0.4/src/auxil/mod.rs new file mode 100644 index 0000000..4d82c1d --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/auxil/mod.rs @@ -0,0 +1,231 @@ +#[cfg(dx12)] +pub(super) mod dxgi; + +#[cfg(all(native, feature = "renderdoc"))] +pub(super) mod renderdoc; + +pub mod db { + pub mod amd { + /// cbindgen:ignore + pub const VENDOR: u32 = 0x1002; + } + pub mod apple { + /// cbindgen:ignore + pub const VENDOR: u32 = 0x106B; + } + pub mod arm { + /// cbindgen:ignore + pub const VENDOR: u32 = 0x13B5; + } + pub mod broadcom { + /// cbindgen:ignore + pub const VENDOR: u32 = 0x14E4; + } + pub mod imgtec { + /// cbindgen:ignore + pub const VENDOR: u32 = 0x1010; + } + pub mod intel { + /// cbindgen:ignore + pub const VENDOR: u32 = 0x8086; + pub const DEVICE_KABY_LAKE_MASK: u32 = 0x5900; + pub const DEVICE_SKY_LAKE_MASK: u32 = 0x1900; + } + pub mod mesa { + // Mesa does not actually have a PCI vendor id. + // + // To match Vulkan, we use the VkVendorId for Mesa in the gles backend so that lavapipe (Vulkan) and + // llvmpipe (OpenGL) have the same vendor id. + /// cbindgen:ignore + pub const VENDOR: u32 = 0x10005; + } + pub mod nvidia { + /// cbindgen:ignore + pub const VENDOR: u32 = 0x10DE; + } + pub mod qualcomm { + /// cbindgen:ignore + pub const VENDOR: u32 = 0x5143; + } +} + +/// Maximum binding size for the shaders that only support `i32` indexing. +/// Interestingly, the index itself can't reach that high, because the minimum +/// element size is 4 bytes, but the compiler toolchain still computes the +/// offset at some intermediate point, internally, as i32. +pub const MAX_I32_BINDING_SIZE: u32 = (1 << 31) - 1; + +pub use wgpu_naga_bridge::map_naga_stage; + +impl crate::CopyExtent { + pub fn map_extent_to_copy_size(extent: &wgt::Extent3d, dim: wgt::TextureDimension) -> Self { + Self { + width: extent.width, + height: extent.height, + depth: match dim { + wgt::TextureDimension::D1 | wgt::TextureDimension::D2 => 1, + wgt::TextureDimension::D3 => extent.depth_or_array_layers, + }, + } + } + + pub fn min(&self, other: &Self) -> Self { + Self { + width: self.width.min(other.width), + height: self.height.min(other.height), + depth: self.depth.min(other.depth), + } + } + + // Get the copy size at a specific mipmap level. This doesn't make most sense, + // since the copy extents are provided *for* a mipmap level to start with. + // But backends use `CopyExtent` more sparingly, and this piece is shared. + pub fn at_mip_level(&self, level: u32) -> Self { + Self { + width: (self.width >> level).max(1), + height: (self.height >> level).max(1), + depth: (self.depth >> level).max(1), + } + } +} + +impl crate::TextureCopyBase { + pub fn max_copy_size(&self, full_size: &crate::CopyExtent) -> crate::CopyExtent { + let mip = full_size.at_mip_level(self.mip_level); + crate::CopyExtent { + width: mip.width - self.origin.x, + height: mip.height - self.origin.y, + depth: mip.depth - self.origin.z, + } + } +} + +impl crate::BufferTextureCopy { + pub fn clamp_size_to_virtual(&mut self, full_size: &crate::CopyExtent) { + let max_size = self.texture_base.max_copy_size(full_size); + self.size = self.size.min(&max_size); + } +} + +impl crate::TextureCopy { + pub fn clamp_size_to_virtual( + &mut self, + full_src_size: &crate::CopyExtent, + full_dst_size: &crate::CopyExtent, + ) { + let max_src_size = self.src_base.max_copy_size(full_src_size); + let max_dst_size = self.dst_base.max_copy_size(full_dst_size); + self.size = self.size.min(&max_src_size).min(&max_dst_size); + } +} + +/// Adjust `limits` to honor HAL-imposed maximums and comply with WebGPU's +/// adapter capability guarantees. +#[cfg_attr(any(not(any_backend), metal), allow(dead_code))] +pub(crate) fn adjust_raw_limits(mut limits: wgt::Limits) -> wgt::Limits { + // Apply hal limits. + limits.max_bind_groups = limits.max_bind_groups.min(crate::MAX_BIND_GROUPS as u32); + limits.max_vertex_buffers = limits + .max_vertex_buffers + .min(crate::MAX_VERTEX_BUFFERS as u32); + limits.max_color_attachments = limits + .max_color_attachments + .min(crate::MAX_COLOR_ATTACHMENTS as u32); + + // Adjust limits according to WebGPU adapter capability guarantees. + // See . + + // WebGPU requires maxBindingsPerBindGroup to be at least the sum of all + // per-stage limits multiplied with the maximum shader stages per pipeline. + // + // Since backends already report their maximum maxBindingsPerBindGroup, + // we need to lower all per-stage limits to satisfy this guarantee. + const MAX_SHADER_STAGES_PER_PIPELINE: u32 = 2; + let max_per_stage_resources = + limits.max_bindings_per_bind_group / MAX_SHADER_STAGES_PER_PIPELINE; + + cap_limits_to_be_under_the_sum_limit( + [ + &mut limits.max_sampled_textures_per_shader_stage, + &mut limits.max_uniform_buffers_per_shader_stage, + &mut limits.max_storage_textures_per_shader_stage, + &mut limits.max_storage_buffers_per_shader_stage, + &mut limits.max_samplers_per_shader_stage, + &mut limits.max_acceleration_structures_per_shader_stage, + ], + max_per_stage_resources, + ); + + // Not required by the spec but dynamic buffers count + // towards non-dynamic buffer limits as well. + limits.max_dynamic_uniform_buffers_per_pipeline_layout = limits + .max_dynamic_uniform_buffers_per_pipeline_layout + .min(limits.max_uniform_buffers_per_shader_stage); + limits.max_dynamic_storage_buffers_per_pipeline_layout = limits + .max_dynamic_storage_buffers_per_pipeline_layout + .min(limits.max_storage_buffers_per_shader_stage); + + limits.min_uniform_buffer_offset_alignment = limits.min_uniform_buffer_offset_alignment.max(32); + limits.min_storage_buffer_offset_alignment = limits.min_storage_buffer_offset_alignment.max(32); + + limits.max_uniform_buffer_binding_size = limits + .max_uniform_buffer_binding_size + .min(limits.max_buffer_size); + limits.max_storage_buffer_binding_size = limits + .max_storage_buffer_binding_size + .min(limits.max_buffer_size); + + limits.max_storage_buffer_binding_size &= !(u64::from(wgt::STORAGE_BINDING_SIZE_ALIGNMENT) - 1); + limits.max_vertex_buffer_array_stride &= !(wgt::VERTEX_ALIGNMENT as u32 - 1); + + let x = limits.max_compute_workgroup_size_x; + let y = limits.max_compute_workgroup_size_y; + let z = limits.max_compute_workgroup_size_z; + let m = limits.max_compute_invocations_per_workgroup; + limits.max_compute_workgroup_size_x = x.min(m); + limits.max_compute_workgroup_size_y = y.min(m); + limits.max_compute_workgroup_size_z = z.min(m); + limits.max_compute_invocations_per_workgroup = m.min(x.saturating_mul(y).saturating_mul(z)); + + limits +} + +/// Evenly allocates space to each limit, +/// capping them only if strictly necessary. +pub fn cap_limits_to_be_under_the_sum_limit( + mut limits: [&mut u32; N], + sum_limit: u32, +) { + limits.sort(); + + let mut rem_limit = sum_limit; + let mut divisor = limits.len() as u32; + for limit_to_adjust in limits { + let limit = rem_limit / divisor; + *limit_to_adjust = (*limit_to_adjust).min(limit); + rem_limit -= *limit_to_adjust; + divisor -= 1; + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_cap_limits_to_be_under_the_sum_limit() { + test([3, 3, 3], 3, [1, 1, 1]); + test([3, 2, 1], 3, [1, 1, 1]); + test([1, 2, 3], 6, [1, 2, 3]); + test([1, 2, 3], 3, [1, 1, 1]); + test([1, 8, 100], 6, [1, 2, 3]); + test([2, 80, 80], 6, [2, 2, 2]); + test([2, 80, 80], 12, [2, 5, 5]); + + #[track_caller] + fn test(mut input: [u32; N], limit: u32, output: [u32; N]) { + cap_limits_to_be_under_the_sum_limit(input.each_mut(), limit); + assert_eq!(input, output); + } + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/auxil/renderdoc.rs b/third_party/wgpu-hal-29.0.4/src/auxil/renderdoc.rs new file mode 100644 index 0000000..52fa2b0 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/auxil/renderdoc.rs @@ -0,0 +1,141 @@ +//! RenderDoc integration - +#![cfg_attr(not(any(feature = "gles", feature = "vulkan")), allow(dead_code))] + +use alloc::format; +use alloc::string::String; +use core::{ffi, ptr}; + +/// The dynamically loaded RenderDoc API function table +#[repr(C)] +#[derive(Debug)] +pub struct RenderDocApi { + api: renderdoc_sys::RENDERDOC_API_1_4_1, + lib: libloading::Library, +} + +unsafe impl Send for RenderDocApi {} +unsafe impl Sync for RenderDocApi {} + +/// RenderDoc API type +#[derive(Debug)] +pub enum RenderDoc { + /// RenderDoc functionality is available + Available { + /// RenderDoc API with function pointers + api: RenderDocApi, + }, + /// RenderDoc functionality is _not_ available + NotAvailable { + /// A description why renderdoc functionality is not available + reason: String, + }, +} + +// TODO: replace with libloading API once supported +#[cfg(unix)] +const RTLD_NOLOAD: i32 = 0x4; + +impl RenderDoc { + pub unsafe fn new() -> Self { + type GetApiFn = unsafe extern "C" fn(version: u32, out: *mut *mut ffi::c_void) -> i32; + + #[cfg(windows)] + let renderdoc_filename = "renderdoc.dll"; + #[cfg(all(unix, not(target_os = "android")))] + let renderdoc_filename = "librenderdoc.so"; + #[cfg(target_os = "android")] + let renderdoc_filename = "libVkLayer_GLES_RenderDoc.so"; + + #[cfg(unix)] + let renderdoc_result: Result = unsafe { + libloading::os::unix::Library::open( + Some(renderdoc_filename), + libloading::os::unix::RTLD_NOW | RTLD_NOLOAD, + ) + } + .map(|lib| lib.into()); + + #[cfg(windows)] + let renderdoc_result: Result = + libloading::os::windows::Library::open_already_loaded(renderdoc_filename) + .map(|lib| lib.into()); + + let renderdoc_lib = match renderdoc_result { + Ok(lib) => lib, + Err(e) => { + return RenderDoc::NotAvailable { + reason: format!( + "Unable to load renderdoc library '{renderdoc_filename}': {e:?}" + ), + } + } + }; + + let get_api: libloading::Symbol = + match unsafe { renderdoc_lib.get(c"RENDERDOC_GetAPI".to_bytes()) } { + Ok(api) => api, + Err(e) => { + return RenderDoc::NotAvailable { + reason: format!( + "Unable to get RENDERDOC_GetAPI from renderdoc library '{renderdoc_filename}': {e:?}" + ), + } + } + }; + let mut obj = ptr::null_mut(); + match unsafe { get_api(10401, &mut obj) } { + 1 => RenderDoc::Available { + api: RenderDocApi { + api: unsafe { *obj.cast::() }, + lib: renderdoc_lib, + }, + }, + return_value => RenderDoc::NotAvailable { + reason: format!( + "Unable to get API from renderdoc library '{renderdoc_filename}': {return_value}" + ), + }, + } + } +} + +impl Default for RenderDoc { + fn default() -> Self { + if !cfg!(debug_assertions) { + return RenderDoc::NotAvailable { + reason: "RenderDoc support is only enabled with 'debug_assertions'".into(), + }; + } + unsafe { Self::new() } + } +} +/// An implementation specific handle +pub type Handle = *mut ffi::c_void; + +impl RenderDoc { + /// Start a RenderDoc frame capture + pub unsafe fn start_frame_capture(&self, device_handle: Handle, window_handle: Handle) -> bool { + match *self { + Self::Available { api: ref entry } => { + unsafe { entry.api.StartFrameCapture.unwrap()(device_handle, window_handle) }; + true + } + Self::NotAvailable { ref reason } => { + log::warn!("Could not start RenderDoc frame capture: {reason}"); + false + } + } + } + + /// End a RenderDoc frame capture + pub unsafe fn end_frame_capture(&self, device_handle: Handle, window_handle: Handle) { + match *self { + Self::Available { api: ref entry } => { + unsafe { entry.api.EndFrameCapture.unwrap()(device_handle, window_handle) }; + } + Self::NotAvailable { ref reason } => { + log::warn!("Could not end RenderDoc frame capture: {reason}") + } + }; + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/adapter.rs b/third_party/wgpu-hal-29.0.4/src/dx12/adapter.rs new file mode 100644 index 0000000..a86c337 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/adapter.rs @@ -0,0 +1,1472 @@ +use alloc::{string::String, sync::Arc, vec::Vec}; +use core::ptr; +use std::thread; + +use parking_lot::Mutex; +use windows::{ + core::Interface as _, + Win32::{ + Devices::DeviceAndDriverInstallation::{ + SetupDiDestroyDeviceInfoList, SetupDiEnumDeviceInfo, SetupDiGetClassDevsW, + SetupDiGetDeviceRegistryPropertyW, DIGCF_PRESENT, GUID_DEVCLASS_DISPLAY, HDEVINFO, + SPDRP_ADDRESS, SPDRP_BUSNUMBER, SPDRP_HARDWAREID, SP_DEVINFO_DATA, + }, + Foundation::{GetLastError, ERROR_NO_MORE_ITEMS}, + Graphics::{Direct3D, Direct3D12, Dxgi}, + UI::WindowsAndMessaging, + }, +}; + +use super::D3D12Lib; +use crate::{ + auxil::{ + self, + dxgi::{factory::DxgiAdapter, result::HResult}, + }, + dx12::{ + dcomp::DCompLib, device_creation::DeviceFactory, shader_compilation, FeatureLevel, + ShaderModel, SurfaceTarget, + }, +}; + +impl Drop for super::Adapter { + fn drop(&mut self) { + // Debug tracking alive objects + if !thread::panicking() + && self + .private_caps + .instance_flags + .contains(wgt::InstanceFlags::VALIDATION) + { + unsafe { + self.report_live_objects(); + } + } + } +} + +impl super::Adapter { + pub unsafe fn report_live_objects(&self) { + if let Ok(debug_device) = self.raw.cast::() { + unsafe { + debug_device.ReportLiveDeviceObjects( + Direct3D12::D3D12_RLDO_SUMMARY | Direct3D12::D3D12_RLDO_IGNORE_INTERNAL, + ) + } + .unwrap() + } + } + + pub fn raw_adapter(&self) -> &DxgiAdapter { + &self.raw + } + + #[allow(clippy::too_many_arguments)] + pub(super) fn expose( + adapter: DxgiAdapter, + library: &Arc, + device_factory: &Arc, + dcomp_lib: &Arc, + instance_flags: wgt::InstanceFlags, + memory_budget_thresholds: wgt::MemoryBudgetThresholds, + compiler_container: Arc, + backend_options: wgt::Dx12BackendOptions, + telemetry: Option, + ) -> Option> { + let desc = unsafe { adapter.GetDesc2() }.unwrap(); + let driver_version = unsafe { adapter.CheckInterfaceSupport(&Dxgi::IDXGIDevice::IID) }; + let driver_version = driver_version + .map(|driver_version| { + let driver_version = driver_version as u64; + [ + (driver_version >> 48) as u16, + (driver_version >> 32) as u16, + (driver_version >> 16) as u16, + driver_version as u16, + ] + }) + .map_err(|e| e.code()); + + // Create the device so that we can get the capabilities. + let res = { + profiling::scope!("ID3D12Device::create_device"); + device_factory.create_device(library, &adapter, Direct3D::D3D_FEATURE_LEVEL_11_0) + }; + if let Some(telemetry) = telemetry { + if let Err(err) = res { + (telemetry.d3d12_expose_adapter)( + &desc, + driver_version, + crate::D3D12ExposeAdapterResult::CreateDeviceError(err), + ); + } + } + let device = res.ok()?; + + profiling::scope!("feature queries"); + + // Detect the highest supported feature level. + let d3d_feature_level = [ + Direct3D::D3D_FEATURE_LEVEL_12_2, + Direct3D::D3D_FEATURE_LEVEL_12_1, + Direct3D::D3D_FEATURE_LEVEL_12_0, + Direct3D::D3D_FEATURE_LEVEL_11_1, + Direct3D::D3D_FEATURE_LEVEL_11_0, + ]; + let mut device_levels = Direct3D12::D3D12_FEATURE_DATA_FEATURE_LEVELS { + NumFeatureLevels: d3d_feature_level.len() as u32, + pFeatureLevelsRequested: d3d_feature_level.as_ptr().cast(), + MaxSupportedFeatureLevel: Default::default(), + }; + unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_FEATURE_LEVELS, + <*mut _>::cast(&mut device_levels), + size_of_val(&device_levels) as u32, + ) + } + .unwrap(); + let max_feature_level = match device_levels.MaxSupportedFeatureLevel { + Direct3D::D3D_FEATURE_LEVEL_11_0 => FeatureLevel::_11_0, + Direct3D::D3D_FEATURE_LEVEL_11_1 => FeatureLevel::_11_1, + Direct3D::D3D_FEATURE_LEVEL_12_0 => FeatureLevel::_12_0, + Direct3D::D3D_FEATURE_LEVEL_12_1 => FeatureLevel::_12_1, + Direct3D::D3D_FEATURE_LEVEL_12_2 => FeatureLevel::_12_2, + fl => { + if let Some(telemetry) = telemetry { + (telemetry.d3d12_expose_adapter)( + &desc, + driver_version, + crate::D3D12ExposeAdapterResult::UnknownFeatureLevel(fl.0), + ); + } + return None; + } + }; + + let device_name = auxil::dxgi::conv::map_adapter_name(desc.Description); + + let mut features_architecture = Direct3D12::D3D12_FEATURE_DATA_ARCHITECTURE::default(); + + unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_ARCHITECTURE, + <*mut _>::cast(&mut features_architecture), + size_of_val(&features_architecture) as u32, + ) + } + .unwrap(); + + let mut features1 = Direct3D12::D3D12_FEATURE_DATA_D3D12_OPTIONS1::default(); + let hr = unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_D3D12_OPTIONS1, + <*mut _>::cast(&mut features1), + size_of_val(&features1) as u32, + ) + }; + + let mut workarounds = super::Workarounds::default(); + + let is_warp = device_name.contains("Microsoft Basic Render Driver"); + + // WARP uses two different versioning schemes. Versions that ship with windows + // use a version that starts with 10.x.x.x. Versions that ship from Nuget use 1.0.x.x. + // + // As far as we know, this is only an issue on the Nuget versions. + if let Ok(driver_version) = driver_version { + if is_warp && driver_version >= [1, 0, 13, 0] && driver_version[0] < 10 { + workarounds.avoid_shader_debug_info = true; + } + } + + let driver_version_string = { + let driver_version = driver_version.unwrap_or([0, 0, 0, 0]); + format!( + "{}.{}.{}.{}", + driver_version[0], driver_version[1], driver_version[2], driver_version[3] + ) + }; + + let info = wgt::AdapterInfo { + backend: wgt::Backend::Dx12, + name: device_name, + vendor: desc.VendorId, + device: desc.DeviceId, + device_type: if Dxgi::DXGI_ADAPTER_FLAG(desc.Flags as i32) + .contains(Dxgi::DXGI_ADAPTER_FLAG_SOFTWARE) + { + wgt::DeviceType::Cpu + } else if features_architecture.UMA.as_bool() { + wgt::DeviceType::IntegratedGpu + } else { + wgt::DeviceType::DiscreteGpu + }, + device_pci_bus_id: get_adapter_pci_info(desc.VendorId, desc.DeviceId), + driver: driver_version_string, + driver_info: String::new(), + subgroup_min_size: features1.WaveLaneCountMin, + subgroup_max_size: features1.WaveLaneCountMax, + transient_saves_memory: false, + }; + + let mut options = Direct3D12::D3D12_FEATURE_DATA_D3D12_OPTIONS::default(); + unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_D3D12_OPTIONS, + <*mut _>::cast(&mut options), + size_of_val(&options) as u32, + ) + } + .unwrap(); + + /// Resource Binding Tiers: https://learn.microsoft.com/en-us/windows/win32/direct3d12/hardware-support#limits-dependant-on-hardware + #[derive(PartialEq, Eq, PartialOrd, Ord)] + enum ResourceBindingTier { + T1, + T2, + T3, + } + let rbt = match options.ResourceBindingTier { + Direct3D12::D3D12_RESOURCE_BINDING_TIER_1 => ResourceBindingTier::T1, + Direct3D12::D3D12_RESOURCE_BINDING_TIER_2 => ResourceBindingTier::T2, + tier if tier.0 >= Direct3D12::D3D12_RESOURCE_BINDING_TIER_3.0 => { + ResourceBindingTier::T3 + } + other => { + log::debug!("Got zero or negative value for resource binding tier {other:?}"); + ResourceBindingTier::T1 + } + }; + + if rbt == ResourceBindingTier::T1 { + if let Some(telemetry) = telemetry { + (telemetry.d3d12_expose_adapter)( + &desc, + driver_version, + crate::D3D12ExposeAdapterResult::ResourceBindingTier2Requirement, + ); + } + // We require Tier 2 or higher for the ability to make samplers bindless in all cases. + return None; + } + + let _depth_bounds_test_supported = { + let mut features2 = Direct3D12::D3D12_FEATURE_DATA_D3D12_OPTIONS2::default(); + unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_D3D12_OPTIONS2, + <*mut _>::cast(&mut features2), + size_of_val(&features2) as u32, + ) + } + .is_ok() + && features2.DepthBoundsTestSupported.as_bool() + }; + + let (casting_fully_typed_format_supported, view_instancing) = { + let mut features3 = Direct3D12::D3D12_FEATURE_DATA_D3D12_OPTIONS3::default(); + if unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_D3D12_OPTIONS3, + <*mut _>::cast(&mut features3), + size_of_val(&features3) as u32, + ) + } + .is_ok() + { + ( + features3.CastingFullyTypedFormatSupported.as_bool(), + features3.ViewInstancingTier.0 >= Direct3D12::D3D12_VIEW_INSTANCING_TIER_1.0, + ) + } else { + (false, false) + } + }; + + let heap_create_not_zeroed = { + // For D3D12_HEAP_FLAG_CREATE_NOT_ZEROED we just need to + // make sure that options7 can be queried. See also: + // https://devblogs.microsoft.com/directx/coming-to-directx-12-more-control-over-memory-allocation/ + let mut features7 = Direct3D12::D3D12_FEATURE_DATA_D3D12_OPTIONS7::default(); + unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_D3D12_OPTIONS7, + <*mut _>::cast(&mut features7), + size_of_val(&features7) as u32, + ) + } + .is_ok() + }; + + let unrestricted_buffer_texture_copy_pitch_supported = { + let mut features13 = Direct3D12::D3D12_FEATURE_DATA_D3D12_OPTIONS13::default(); + unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_D3D12_OPTIONS13, + <*mut _>::cast(&mut features13), + size_of_val(&features13) as u32, + ) + } + .is_ok() + && features13 + .UnrestrictedBufferTextureCopyPitchSupported + .as_bool() + }; + + let mut max_sampler_descriptor_heap_size = + Direct3D12::D3D12_MAX_SHADER_VISIBLE_SAMPLER_HEAP_SIZE; + { + let mut features19 = Direct3D12::D3D12_FEATURE_DATA_D3D12_OPTIONS19::default(); + let res = unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_D3D12_OPTIONS19, + <*mut _>::cast(&mut features19), + size_of_val(&features19) as u32, + ) + }; + + // Sometimes on Windows 11 23H2, the function returns success, even though the runtime + // does not know about `Options19`. This can cause this number to be 0 as the structure isn't written to. + // This value is nonsense and creating zero-sized sampler heaps can cause drivers to explode. + // As as we're guaranteed 2048 anyway, we make sure this value is not under 2048. + // + // https://github.com/gfx-rs/wgpu/issues/7053 + let is_ok = res.is_ok(); + let is_above_minimum = features19.MaxSamplerDescriptorHeapSize + > Direct3D12::D3D12_MAX_SHADER_VISIBLE_SAMPLER_HEAP_SIZE; + if is_ok && is_above_minimum { + max_sampler_descriptor_heap_size = features19.MaxSamplerDescriptorHeapSize; + } + }; + + let mut shader_models_after_5_1 = [ + Direct3D12::D3D_SHADER_MODEL_6_9, + Direct3D12::D3D_SHADER_MODEL_6_8, + Direct3D12::D3D_SHADER_MODEL_6_7, + Direct3D12::D3D_SHADER_MODEL_6_6, + Direct3D12::D3D_SHADER_MODEL_6_5, + Direct3D12::D3D_SHADER_MODEL_6_4, + Direct3D12::D3D_SHADER_MODEL_6_3, + Direct3D12::D3D_SHADER_MODEL_6_2, + Direct3D12::D3D_SHADER_MODEL_6_1, + Direct3D12::D3D_SHADER_MODEL_6_0, + ] + .iter(); + let max_device_shader_model = loop { + if let Some(&sm) = shader_models_after_5_1.next() { + let mut sm = Direct3D12::D3D12_FEATURE_DATA_SHADER_MODEL { + HighestShaderModel: sm, + }; + if unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_SHADER_MODEL, + <*mut _>::cast(&mut sm), + size_of_val(&sm) as u32, + ) + } + .is_ok() + { + break match sm.HighestShaderModel { + Direct3D12::D3D_SHADER_MODEL_5_1 => ShaderModel::_5_1, + Direct3D12::D3D_SHADER_MODEL_6_0 => ShaderModel::_6_0, + Direct3D12::D3D_SHADER_MODEL_6_1 => ShaderModel::_6_1, + Direct3D12::D3D_SHADER_MODEL_6_2 => ShaderModel::_6_2, + Direct3D12::D3D_SHADER_MODEL_6_3 => ShaderModel::_6_3, + Direct3D12::D3D_SHADER_MODEL_6_4 => ShaderModel::_6_4, + Direct3D12::D3D_SHADER_MODEL_6_5 => ShaderModel::_6_5, + Direct3D12::D3D_SHADER_MODEL_6_6 => ShaderModel::_6_6, + Direct3D12::D3D_SHADER_MODEL_6_7 => ShaderModel::_6_7, + Direct3D12::D3D_SHADER_MODEL_6_8 => ShaderModel::_6_8, + Direct3D12::D3D_SHADER_MODEL_6_9 => ShaderModel::_6_9, + _ => unreachable!(), + }; + } + } else { + break ShaderModel::_5_1; + } + }; + + let wgt_shader_model = backend_options + .force_shader_model + .get() + .or(compiler_container.max_shader_model()); + + let shader_model = if let Some(max_shader_model) = wgt_shader_model { + let max_dxc_shader_model = match max_shader_model { + wgt::DxcShaderModel::V6_0 => ShaderModel::_6_0, + wgt::DxcShaderModel::V6_1 => ShaderModel::_6_1, + wgt::DxcShaderModel::V6_2 => ShaderModel::_6_2, + wgt::DxcShaderModel::V6_3 => ShaderModel::_6_3, + wgt::DxcShaderModel::V6_4 => ShaderModel::_6_4, + wgt::DxcShaderModel::V6_5 => ShaderModel::_6_5, + wgt::DxcShaderModel::V6_6 => ShaderModel::_6_6, + wgt::DxcShaderModel::V6_7 => ShaderModel::_6_7, + wgt::DxcShaderModel::V6_8 => ShaderModel::_6_8, + wgt::DxcShaderModel::V6_9 => ShaderModel::_6_9, + }; + + let shader_model = max_device_shader_model.min(max_dxc_shader_model); + + match shader_model { + ShaderModel::_5_1 => { + if let Some(telemetry) = telemetry { + (telemetry.d3d12_expose_adapter)( + &desc, + driver_version, + crate::D3D12ExposeAdapterResult::ShaderModel6Requirement, + ); + } + // don't expose this adapter if it doesn't support DXIL + return None; + } + ShaderModel::_6_0 => naga::back::hlsl::ShaderModel::V6_0, + ShaderModel::_6_1 => naga::back::hlsl::ShaderModel::V6_1, + ShaderModel::_6_2 => naga::back::hlsl::ShaderModel::V6_2, + ShaderModel::_6_3 => naga::back::hlsl::ShaderModel::V6_3, + ShaderModel::_6_4 => naga::back::hlsl::ShaderModel::V6_4, + ShaderModel::_6_5 => naga::back::hlsl::ShaderModel::V6_5, + ShaderModel::_6_6 => naga::back::hlsl::ShaderModel::V6_6, + ShaderModel::_6_7 => naga::back::hlsl::ShaderModel::V6_7, + ShaderModel::_6_8 => naga::back::hlsl::ShaderModel::V6_8, + ShaderModel::_6_9 => naga::back::hlsl::ShaderModel::V6_9, + } + } else { + naga::back::hlsl::ShaderModel::V5_1 + }; + let private_caps = super::PrivateCapabilities { + instance_flags, + workarounds, + heterogeneous_resource_heaps: options.ResourceHeapTier + != Direct3D12::D3D12_RESOURCE_HEAP_TIER_1, + memory_architecture: if features_architecture.UMA.as_bool() { + super::MemoryArchitecture::Unified { + cache_coherent: features_architecture.CacheCoherentUMA.as_bool(), + } + } else { + super::MemoryArchitecture::NonUnified + }, + heap_create_not_zeroed, + casting_fully_typed_format_supported, + // See https://github.com/gfx-rs/wgpu/issues/3552 + suballocation_supported: !info.name.contains("Iris(R) Xe"), + shader_model, + max_sampler_descriptor_heap_size, + unrestricted_buffer_texture_copy_pitch_supported, + }; + + // these should always be available on d3d12 + let mut features = wgt::Features::empty() + | wgt::Features::DEPTH_CLIP_CONTROL + | wgt::Features::DEPTH32FLOAT_STENCIL8 + | wgt::Features::INDIRECT_FIRST_INSTANCE + | wgt::Features::MAPPABLE_PRIMARY_BUFFERS + | wgt::Features::MULTI_DRAW_INDIRECT_COUNT + | wgt::Features::ADDRESS_MODE_CLAMP_TO_BORDER + | wgt::Features::ADDRESS_MODE_CLAMP_TO_ZERO + | wgt::Features::POLYGON_MODE_LINE + | wgt::Features::TEXTURE_ADAPTER_SPECIFIC_FORMAT_FEATURES + | wgt::Features::TIMESTAMP_QUERY + | wgt::Features::TIMESTAMP_QUERY_INSIDE_ENCODERS + | wgt::Features::TIMESTAMP_QUERY_INSIDE_PASSES + | wgt::Features::TEXTURE_COMPRESSION_BC + | wgt::Features::TEXTURE_COMPRESSION_BC_SLICED_3D + | wgt::Features::CLEAR_TEXTURE + | wgt::Features::TEXTURE_FORMAT_16BIT_NORM + | wgt::Features::IMMEDIATES + | wgt::Features::PRIMITIVE_INDEX + | wgt::Features::RG11B10UFLOAT_RENDERABLE + | wgt::Features::DUAL_SOURCE_BLENDING + | wgt::Features::TEXTURE_FORMAT_NV12 + | wgt::Features::FLOAT32_FILTERABLE + | wgt::Features::TEXTURE_ATOMIC + | wgt::Features::PASSTHROUGH_SHADERS + | wgt::Features::EXTERNAL_TEXTURE + | wgt::Features::MEMORY_DECORATION_COHERENT; + + //TODO: in order to expose this, we need to run a compute shader + // that extract the necessary statistics out of the D3D12 result. + // Alternatively, we could allocate a buffer for the query set, + // write the results there, and issue a bunch of copy commands. + //| wgt::Features::PIPELINE_STATISTICS_QUERY + + if max_feature_level >= FeatureLevel::_11_1 { + features |= wgt::Features::VERTEX_WRITABLE_STORAGE; + } + + features.set( + wgt::Features::CONSERVATIVE_RASTERIZATION, + options.ConservativeRasterizationTier + != Direct3D12::D3D12_CONSERVATIVE_RASTERIZATION_TIER_NOT_SUPPORTED, + ); + + features.set( + wgt::Features::TEXTURE_BINDING_ARRAY + | wgt::Features::STORAGE_RESOURCE_BINDING_ARRAY + | wgt::Features::STORAGE_TEXTURE_ARRAY_NON_UNIFORM_INDEXING + | wgt::Features::SAMPLED_TEXTURE_AND_STORAGE_BUFFER_ARRAY_NON_UNIFORM_INDEXING + // See note below the table https://learn.microsoft.com/en-us/windows/win32/direct3d12/hardware-support + | wgt::Features::PARTIALLY_BOUND_BINDING_ARRAY, + shader_model >= naga::back::hlsl::ShaderModel::V5_1 && rbt >= ResourceBindingTier::T3, + ); + + let bgra8unorm_storage_supported = { + let mut bgra8unorm_info = Direct3D12::D3D12_FEATURE_DATA_FORMAT_SUPPORT { + Format: Dxgi::Common::DXGI_FORMAT_B8G8R8A8_UNORM, + ..Default::default() + }; + let hr = unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_FORMAT_SUPPORT, + <*mut _>::cast(&mut bgra8unorm_info), + size_of_val(&bgra8unorm_info) as u32, + ) + }; + hr.is_ok() + && bgra8unorm_info + .Support2 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT2_UAV_TYPED_STORE) + }; + features.set( + wgt::Features::BGRA8UNORM_STORAGE, + bgra8unorm_storage_supported, + ); + + let p010_format_supported = { + let mut p010_info = Direct3D12::D3D12_FEATURE_DATA_FORMAT_SUPPORT { + Format: Dxgi::Common::DXGI_FORMAT_P010, + ..Default::default() + }; + let hr = unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_FORMAT_SUPPORT, + <*mut _>::cast(&mut p010_info), + size_of_val(&p010_info) as u32, + ) + }; + if hr.is_ok() { + let supports_texture2d = p010_info + .Support1 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT1_TEXTURE2D); + let supports_shader_load = p010_info + .Support1 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT1_SHADER_LOAD); + let supports_shader_sample = p010_info + .Support1 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT1_SHADER_SAMPLE); + supports_texture2d && supports_shader_load && supports_shader_sample + } else { + false + } + }; + features.set(wgt::Features::TEXTURE_FORMAT_P010, p010_format_supported); + + features.set( + wgt::Features::SHADER_INT64, + shader_model >= naga::back::hlsl::ShaderModel::V6_0 + && hr.is_ok() + && features1.Int64ShaderOps.as_bool(), + ); + + let float16_supported = { + let mut features4 = Direct3D12::D3D12_FEATURE_DATA_D3D12_OPTIONS4::default(); + let hr = unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_D3D12_OPTIONS4, // https://learn.microsoft.com/en-us/windows/win32/api/d3d12/ne-d3d12-d3d12_feature#syntax + ptr::from_mut(&mut features4).cast(), + size_of::() as _, + ) + }; + hr.is_ok() && features4.Native16BitShaderOpsSupported.as_bool() + }; + + features.set( + wgt::Features::SHADER_F16, + shader_model >= naga::back::hlsl::ShaderModel::V6_2 && float16_supported, + ); + + features.set( + wgt::Features::SUBGROUP, + shader_model >= naga::back::hlsl::ShaderModel::V6_0 + && hr.is_ok() + && features1.WaveOps.as_bool(), + ); + let mut features5 = Direct3D12::D3D12_FEATURE_DATA_D3D12_OPTIONS5::default(); + let has_features5 = unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_D3D12_OPTIONS5, + <*mut _>::cast(&mut features5), + size_of_val(&features5) as u32, + ) + } + .is_ok(); + + // Once ray tracing pipelines are supported they also will go here + let supports_ray_tracing = features5.RaytracingTier.0 + >= Direct3D12::D3D12_RAYTRACING_TIER_1_1.0 + && shader_model >= naga::back::hlsl::ShaderModel::V6_5 + && has_features5; + + features.set( + wgt::Features::EXPERIMENTAL_RAY_QUERY + | wgt::Features::EXTENDED_ACCELERATION_STRUCTURE_VERTEX_FORMATS, + supports_ray_tracing, + ); + + // Binding arrays of TLAS are supported on D3D12 when ray tracing is supported. + // + // This flag is used for shader-side `binding_array` as well as + // allowing `BindGroupLayoutEntry::count = Some(...)` for `BindingType::AccelerationStructure`. + features.set( + wgt::Features::ACCELERATION_STRUCTURE_BINDING_ARRAY, + supports_ray_tracing, + ); + + // Check for Int64 atomic support on buffers. This is very convoluted, but is based on a conservative reading + // of https://microsoft.github.io/DirectX-Specs/d3d/HLSL_SM_6_6_Int64_and_Float_Atomics.html#integer-64-bit-capabilities. + let atomic_int64_buffers; + let atomic_int64_textures; + { + let mut features9 = Direct3D12::D3D12_FEATURE_DATA_D3D12_OPTIONS9::default(); + let hr9 = unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_D3D12_OPTIONS9, + <*mut _>::cast(&mut features9), + size_of_val(&features9) as u32, + ) + } + .is_ok(); + + let mut features11 = Direct3D12::D3D12_FEATURE_DATA_D3D12_OPTIONS11::default(); + let hr11 = unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_D3D12_OPTIONS11, + <*mut _>::cast(&mut features11), + size_of_val(&features11) as u32, + ) + } + .is_ok(); + + atomic_int64_buffers = hr9 && hr11 && hr.is_ok() + // Int64 atomics show up in SM6.6. + && shader_model >= naga::back::hlsl::ShaderModel::V6_6 + // They require Int64 to be available in the shader at all. + && features1.Int64ShaderOps.as_bool() + // As our RWByteAddressBuffers can exist on both descriptor heaps and + // as root descriptors, we need to ensure that both cases are supported. + // base SM6.6 only guarantees Int64 atomics on resources in root descriptors. + && features11.AtomicInt64OnDescriptorHeapResourceSupported.as_bool() + // Our Int64 atomic caps currently require groupshared. This + // prevents Intel or Qcomm from using Int64 currently. + // https://github.com/gfx-rs/wgpu/issues/8666 + && features9.AtomicInt64OnGroupSharedSupported.as_bool(); + + atomic_int64_textures = hr9 && hr11 && hr.is_ok() + // Int64 atomics show up in SM6.6. + && shader_model >= naga::back::hlsl::ShaderModel::V6_6 + // They require Int64 to be available in the shader at all. + && features1.Int64ShaderOps.as_bool() + // Textures are typed resources, so we need this flag. + && features9.AtomicInt64OnTypedResourceSupported.as_bool() + // As textures can only exist in descriptor heaps, we require this. + // However, all architectures that support atomics on typed resources + // support this as well, so this is somewhat redundant. + && features11.AtomicInt64OnDescriptorHeapResourceSupported.as_bool(); + }; + features.set( + wgt::Features::SHADER_INT64_ATOMIC_ALL_OPS | wgt::Features::SHADER_INT64_ATOMIC_MIN_MAX, + atomic_int64_buffers, + ); + features.set(wgt::Features::TEXTURE_INT64_ATOMIC, atomic_int64_textures); + let mesh_shader_supported = { + let mut features7 = Direct3D12::D3D12_FEATURE_DATA_D3D12_OPTIONS7::default(); + unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_D3D12_OPTIONS7, + <*mut _>::cast(&mut features7), + size_of_val(&features7) as u32, + ) + } + .is_ok() + && features7.MeshShaderTier != Direct3D12::D3D12_MESH_SHADER_TIER_NOT_SUPPORTED + }; + features.set( + wgt::Features::EXPERIMENTAL_MESH_SHADER, + mesh_shader_supported, + ); + let shader_barycentrics_supported = { + let mut features3 = Direct3D12::D3D12_FEATURE_DATA_D3D12_OPTIONS3::default(); + unsafe { + device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_D3D12_OPTIONS3, + <*mut _>::cast(&mut features3), + size_of_val(&features3) as u32, + ) + } + .is_ok() + && features3.BarycentricsSupported.as_bool() + && shader_model >= naga::back::hlsl::ShaderModel::V6_1 + }; + features.set( + wgt::Features::SHADER_BARYCENTRICS, + shader_barycentrics_supported, + ); + + features.set( + wgt::Features::MULTIVIEW, + view_instancing && shader_model >= naga::back::hlsl::ShaderModel::V6_1, + ); + features.set( + wgt::Features::SELECTIVE_MULTIVIEW, + view_instancing && shader_model >= naga::back::hlsl::ShaderModel::V6_1, + ); + + features.set( + wgt::Features::EXPERIMENTAL_MESH_SHADER_MULTIVIEW, + mesh_shader_supported + && view_instancing + && shader_model >= naga::back::hlsl::ShaderModel::V6_1, + ); + + // TODO: Determine if IPresentationManager is supported + let presentation_timer = auxil::dxgi::time::PresentationTimer::new_dxgi(); + + let downlevel = wgt::DownlevelCapabilities::default(); + + // Limits that must share D3D12's root signature size of + // D3D12_MAX_ROOT_COST 64 DWORDS (256 bytes). + // + // Root constants and root tables use 1 DWORD. + // Root descriptors use 2 DWORDs. + // Source: https://learn.microsoft.com/en-us/windows/win32/direct3d12/root-signature-limits#memory-limits-and-costs + // + // Per pipeline layout: + // - RootElement::Constant, (immediates) 32 root constants + // (bounded by maxImmediateSize) = 32 x 4 bytes = 128 bytes + // - RootElement::SamplerHeap, a root table = 4 bytes + // - RootElement::SpecialConstantBuffer, 3 root constants = 3 x 4 bytes = 12 bytes + // - RootElement::DynamicOffsetsBuffer, a root constant per dynamic storage buffer + // (bounded by maxDynamicStorageBuffersPerPipelineLayout) = 4 x 4 bytes = 16 bytes + // - RootElement::DynamicUniformBuffer, a root descriptor per dynamic uniform buffer + // (bounded by maxDynamicUniformBuffersPerPipelineLayout) = 8 x 8 bytes = 64 bytes + // Per bind group: + // - RootElement::Table, a root table + // (bounded by maxBindGroups) = 8 x 4 bytes = 32 bytes + // + // Source: logic in `create_pipeline_layout` + // + // Total: 128 + 4 + 12 + 16 + 64 + 32 = 256 bytes + // + let max_immediate_size = 128; + let max_bind_groups = 8; + let max_dynamic_uniform_buffers_per_pipeline_layout = 8; + let max_dynamic_storage_buffers_per_pipeline_layout = 4; + + // "Maximum number of descriptors in a Constant Buffer View (CBV), Shader Resource View (SRV), or Unordered Access View(UAV) heap used for rendering" + let full_heap_count = match rbt { + ResourceBindingTier::T1 | ResourceBindingTier::T2 => 1_000_000, + // 1_000_000+ + ResourceBindingTier::T3 => { + // Theoretically vram limited, but in practice 2^20 is the limit + 1 << 20 + } + }; + + // "Maximum number of Constant Buffer Views in all descriptor tables per shader stage" + let max_uniform_buffers_per_shader_stage = match rbt { + ResourceBindingTier::T1 | ResourceBindingTier::T2 => 14, + _ => full_heap_count, + }; + + // "Maximum number of Shader Resource Views in all descriptor tables per shader stage" + let mut max_srv_per_shader_stage = match rbt { + ResourceBindingTier::T1 => 128, + _ => full_heap_count, + }; + + // We use an extra SRV for all samplers in a bind group. + // See comment in `create_pipeline_layout`. + max_srv_per_shader_stage -= max_bind_groups; + + // If we also support acceleration structures these are shared so we must halve it. + // It's unlikely that this affects anything because most devices that support ray tracing + // probably have a higher binding tier than one. + let mut max_sampled_textures_per_shader_stage = if supports_ray_tracing { + max_srv_per_shader_stage / 2 + } else { + max_srv_per_shader_stage + }; + let mut max_acceleration_structures_per_shader_stage = if supports_ray_tracing { + max_srv_per_shader_stage / 2 + } else { + 0 + }; + + // "Maximum number of Unordered Access Views in all descriptor tables across all stages" + let max_uav_across_all_stages = match rbt { + ResourceBindingTier::T1 => match max_feature_level { + FeatureLevel::_11_0 => 8, + _ => 64, + }, + ResourceBindingTier::T2 => 64, + ResourceBindingTier::T3 => full_heap_count, + }; + const MAX_SHADER_STAGES_PER_PIPELINE: u32 = 2; + // We must share the UAV limit across both storage resource limits. + let max_uav_per_shader_stage = max_uav_across_all_stages / MAX_SHADER_STAGES_PER_PIPELINE; + let max_storage_textures_per_shader_stage = max_uav_per_shader_stage / 2; + let mut max_storage_buffers_per_shader_stage = max_uav_per_shader_stage / 2; + + // WebGPU storage buffers count as 1 SRV if they are read-only + // or as 1 UAV if they are read-write. See comment in + // `create_pipeline_layout`. Make sure we don't exceed + // the maximum number of SRVs for the relevant limits. + auxil::cap_limits_to_be_under_the_sum_limit( + [ + &mut max_sampled_textures_per_shader_stage, + &mut max_acceleration_structures_per_shader_stage, + &mut max_storage_buffers_per_shader_stage, + ], + max_srv_per_shader_stage, + ); + + // "Maximum number of Samplers in all descriptor tables per shader stage" + let max_samplers_per_shader_stage = match rbt { + ResourceBindingTier::T1 => 16, + _ => 2048, + }; + + // See https://microsoft.github.io/DirectX-Specs/d3d/ViewInstancing.html#maximum-viewinstancecount + let max_multiview_view_count = if view_instancing { 4 } else { 0 }; + + if let Some(telemetry) = telemetry { + (telemetry.d3d12_expose_adapter)( + &desc, + driver_version, + crate::D3D12ExposeAdapterResult::Success( + max_feature_level, + max_device_shader_model, + ), + ); + } + + Some(crate::ExposedAdapter { + adapter: super::Adapter { + raw: adapter, + device, + library: Arc::clone(library), + dcomp_lib: Arc::clone(dcomp_lib), + private_caps, + presentation_timer, + memory_budget_thresholds, + compiler_container, + options: backend_options, + }, + info, + features, + capabilities: crate::Capabilities { + limits: auxil::adjust_raw_limits(wgt::Limits { + // + // WebGPU LIMITS: + // Based on https://gpuweb.github.io/gpuweb/correspondence/#limits + // + // 16384 + max_texture_dimension_1d: Direct3D12::D3D12_REQ_TEXTURE1D_U_DIMENSION, + // 16384 + max_texture_dimension_2d: Direct3D12::D3D12_REQ_TEXTURE2D_U_OR_V_DIMENSION + .min(Direct3D12::D3D12_REQ_TEXTURECUBE_DIMENSION), + // 2048 + max_texture_dimension_3d: Direct3D12::D3D12_REQ_TEXTURE3D_U_V_OR_W_DIMENSION, + // 2048 + max_texture_array_layers: Direct3D12::D3D12_REQ_TEXTURE2D_ARRAY_AXIS_DIMENSION, + // No real limit. + max_bindings_per_bind_group: u32::MAX, + max_sampled_textures_per_shader_stage, + max_samplers_per_shader_stage, + max_storage_textures_per_shader_stage, + max_storage_buffers_per_shader_stage, + max_uniform_buffers_per_shader_stage, + // See `InputSlot` param docs: https://learn.microsoft.com/en-ca/windows/win32/api/d3d12/ns-d3d12-d3d12_input_element_desc + max_vertex_buffers: 16, + // Dx12 does not expose a maximum buffer size in the API. + // This limit is chosen to avoid potential issues with drivers should they internally + // store buffer sizes using 32 bit ints (a situation we have already encountered with vulkan). + max_buffer_size: i32::MAX as u64, + max_storage_buffer_binding_size: auxil::MAX_I32_BINDING_SIZE as u64, + // 65536 + max_uniform_buffer_binding_size: + Direct3D12::D3D12_REQ_CONSTANT_BUFFER_ELEMENT_COUNT as u64 * 16, + // 254 + min_uniform_buffer_offset_alignment: + Direct3D12::D3D12_CONSTANT_BUFFER_DATA_PLACEMENT_ALIGNMENT, + // 16 + min_storage_buffer_offset_alignment: + Direct3D12::D3D12_RAW_UAV_SRV_BYTE_ALIGNMENT, + // 32 + max_vertex_attributes: Direct3D12::D3D12_IA_VERTEX_INPUT_RESOURCE_SLOT_COUNT, + // 2048 + max_vertex_buffer_array_stride: Direct3D12::D3D12_SO_BUFFER_MAX_STRIDE_IN_BYTES, + // 31 + max_inter_stage_shader_variables: Direct3D12::D3D12_VS_OUTPUT_REGISTER_COUNT + .min(Direct3D12::D3D12_PS_INPUT_REGISTER_COUNT) + - 1, // - 1 for position + max_immediate_size, + max_bind_groups, + max_dynamic_uniform_buffers_per_pipeline_layout, + max_dynamic_storage_buffers_per_pipeline_layout, + // 8 + max_color_attachments: Direct3D12::D3D12_SIMULTANEOUS_RENDER_TARGET_COUNT, + // 128 (No documented limit) + max_color_attachment_bytes_per_sample: + Direct3D12::D3D12_SIMULTANEOUS_RENDER_TARGET_COUNT + * wgt::TextureFormat::MAX_TARGET_PIXEL_BYTE_COST, + // From: https://microsoft.github.io/DirectX-Specs/d3d/archive/D3D11_3_FunctionalSpec.htm#18.6.6%20Inter-Thread%20Data%20Sharing + max_compute_workgroup_storage_size: 32768, + // 1024 + max_compute_invocations_per_workgroup: + Direct3D12::D3D12_CS_THREAD_GROUP_MAX_THREADS_PER_GROUP, + // 1024 + max_compute_workgroup_size_x: Direct3D12::D3D12_CS_THREAD_GROUP_MAX_X, + // 1024 + max_compute_workgroup_size_y: Direct3D12::D3D12_CS_THREAD_GROUP_MAX_Y, + // 64 + max_compute_workgroup_size_z: Direct3D12::D3D12_CS_THREAD_GROUP_MAX_Z, + // 65535 + max_compute_workgroups_per_dimension: + Direct3D12::D3D12_CS_DISPATCH_MAX_THREAD_GROUPS_PER_DIMENSION, + // + // NATIVE (Non-WebGPU) LIMITS: + // + max_non_sampler_bindings: 1_000_000, + + max_binding_array_elements_per_shader_stage: full_heap_count, + max_binding_array_sampler_elements_per_shader_stage: + Direct3D12::D3D12_MAX_SHADER_VISIBLE_SAMPLER_HEAP_SIZE, + + // Source: https://microsoft.github.io/DirectX-Specs/d3d/MeshShader.html#dispatchmesh-api + max_task_mesh_workgroup_total_count: if mesh_shader_supported { + 2u32.pow(22) + } else { + 0 + }, + // Technically it says "64k" but I highly doubt they want 65536 for compute and exactly 64,000 for task workgroups + max_task_mesh_workgroups_per_dimension: if mesh_shader_supported { + Direct3D12::D3D12_CS_DISPATCH_MAX_THREAD_GROUPS_PER_DIMENSION + } else { + 0 + }, + // Assume this inherits from compute shaders + max_task_invocations_per_workgroup: if mesh_shader_supported { + Direct3D12::D3D12_CS_4_X_THREAD_GROUP_MAX_THREADS_PER_GROUP + } else { + 0 + }, + max_task_invocations_per_dimension: if mesh_shader_supported { + Direct3D12::D3D12_CS_THREAD_GROUP_MAX_Z + } else { + 0 + }, + // Source: https://microsoft.github.io/DirectX-Specs/d3d/MeshShader.html#amplification-shader-and-mesh-shader + max_mesh_invocations_per_workgroup: if mesh_shader_supported { 128 } else { 0 }, + max_mesh_invocations_per_dimension: if mesh_shader_supported { 128 } else { 0 }, + + max_task_payload_size: if mesh_shader_supported { 16384 } else { 0 }, + max_mesh_output_vertices: if mesh_shader_supported { 256 } else { 0 }, + max_mesh_output_primitives: if mesh_shader_supported { 256 } else { 0 }, + // Source: https://microsoft.github.io/DirectX-Specs/d3d/MeshShader.html#sv_rendertargetarrayindex-limitations-based-on-queryable-capability + max_mesh_output_layers: if mesh_shader_supported { 8 } else { 0 }, + max_mesh_multiview_view_count: if mesh_shader_supported { + max_multiview_view_count + } else { + 0 + }, + + max_blas_primitive_count: if supports_ray_tracing { + 1 << 29 // 2^29 + } else { + 0 + }, + max_blas_geometry_count: if supports_ray_tracing { + 1 << 24 // 2^24 + } else { + 0 + }, + max_tlas_instance_count: if supports_ray_tracing { + 1 << 24 // 2^24 + } else { + 0 + }, + max_acceleration_structures_per_shader_stage, + max_binding_array_acceleration_structure_elements_per_shader_stage: + max_acceleration_structures_per_shader_stage, + max_multiview_view_count, + }), + alignments: crate::Alignments { + buffer_copy_offset: wgt::BufferSize::new( + Direct3D12::D3D12_TEXTURE_DATA_PLACEMENT_ALIGNMENT as u64, + ) + .unwrap(), + buffer_copy_pitch: wgt::BufferSize::new( + Direct3D12::D3D12_TEXTURE_DATA_PITCH_ALIGNMENT as u64, + ) + .unwrap(), + // Direct3D correctly bounds-checks all array accesses: + // https://microsoft.github.io/DirectX-Specs/d3d/archive/D3D11_3_FunctionalSpec.htm#18.6.8.2%20Device%20Memory%20Reads + uniform_bounds_check_alignment: wgt::BufferSize::new(1).unwrap(), + raw_tlas_instance_size: size_of::(), + ray_tracing_scratch_buffer_alignment: + Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_BYTE_ALIGNMENT, + }, + downlevel, + cooperative_matrix_properties: Vec::new(), + }, + }) + } +} + +impl crate::Adapter for super::Adapter { + type A = super::Api; + + unsafe fn open( + &self, + features: wgt::Features, + limits: &wgt::Limits, + memory_hints: &wgt::MemoryHints, + ) -> Result, crate::DeviceError> { + let queue: Direct3D12::ID3D12CommandQueue = { + profiling::scope!("ID3D12Device::CreateCommandQueue"); + unsafe { + self.device + .CreateCommandQueue(&Direct3D12::D3D12_COMMAND_QUEUE_DESC { + Type: Direct3D12::D3D12_COMMAND_LIST_TYPE_DIRECT, + Priority: Direct3D12::D3D12_COMMAND_QUEUE_PRIORITY_NORMAL.0, + Flags: Direct3D12::D3D12_COMMAND_QUEUE_FLAG_NONE, + NodeMask: 0, + }) + } + .into_device_result("Queue creation")? + }; + + let device = super::Device::new( + self.raw.clone(), + self.device.clone(), + queue.clone(), + features, + limits, + memory_hints, + self.private_caps, + &self.library, + &self.dcomp_lib, + self.memory_budget_thresholds, + self.compiler_container.clone(), + self.options.clone(), + )?; + Ok(crate::OpenDevice { + device, + queue: super::Queue { + raw: queue, + temp_lists: Mutex::new(Vec::new()), + }, + }) + } + + unsafe fn texture_format_capabilities( + &self, + format: wgt::TextureFormat, + ) -> crate::TextureFormatCapabilities { + use crate::TextureFormatCapabilities as Tfc; + + let raw_format = match auxil::dxgi::conv::map_texture_format_failable(format) { + Some(f) => f, + None => return Tfc::empty(), + }; + let srv_uav_format = if format.is_combined_depth_stencil_format() { + auxil::dxgi::conv::map_texture_format_for_srv_uav( + format, + // use the depth aspect here as opposed to stencil since it has more capabilities + crate::FormatAspects::DEPTH, + ) + } else { + auxil::dxgi::conv::map_texture_format_for_srv_uav( + format, + crate::FormatAspects::from(format), + ) + } + .unwrap(); + + let mut data = Direct3D12::D3D12_FEATURE_DATA_FORMAT_SUPPORT { + Format: raw_format, + ..Default::default() + }; + unsafe { + self.device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_FORMAT_SUPPORT, + <*mut _>::cast(&mut data), + size_of_val(&data) as u32, + ) + } + .unwrap(); + + // Because we use a different format for SRV and UAV views of depth textures, we need to check + // the features that use SRV/UAVs using the no-depth format. + let mut data_srv_uav = Direct3D12::D3D12_FEATURE_DATA_FORMAT_SUPPORT { + Format: srv_uav_format, + Support1: Direct3D12::D3D12_FORMAT_SUPPORT1_NONE, + Support2: Direct3D12::D3D12_FORMAT_SUPPORT2_NONE, + }; + if raw_format != srv_uav_format { + // Only-recheck if we're using a different format + unsafe { + self.device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_FORMAT_SUPPORT, + ptr::addr_of_mut!(data_srv_uav).cast(), + size_of::() as u32, + ) + } + .unwrap(); + } else { + // Same format, just copy over. + data_srv_uav = data; + } + + let mut caps = Tfc::COPY_SRC | Tfc::COPY_DST; + // Cannot use the contains() helper, and windows-rs doesn't provide a .intersect() helper + let is_texture = (data.Support1 + & (Direct3D12::D3D12_FORMAT_SUPPORT1_TEXTURE1D + | Direct3D12::D3D12_FORMAT_SUPPORT1_TEXTURE2D + | Direct3D12::D3D12_FORMAT_SUPPORT1_TEXTURE3D + | Direct3D12::D3D12_FORMAT_SUPPORT1_TEXTURECUBE)) + .0 + != 0; + // SRVs use srv_uav_format + caps.set( + Tfc::SAMPLED, + is_texture + && data_srv_uav + .Support1 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT1_SHADER_LOAD), + ); + caps.set( + Tfc::SAMPLED_LINEAR, + data_srv_uav + .Support1 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT1_SHADER_SAMPLE), + ); + caps.set( + Tfc::COLOR_ATTACHMENT, + data.Support1 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT1_RENDER_TARGET), + ); + caps.set( + Tfc::COLOR_ATTACHMENT_BLEND, + data.Support1 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT1_BLENDABLE), + ); + caps.set( + Tfc::DEPTH_STENCIL_ATTACHMENT, + data.Support1 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT1_DEPTH_STENCIL), + ); + // UAVs use srv_uav_format + caps.set( + Tfc::STORAGE_READ_ONLY, + data_srv_uav + .Support2 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT2_UAV_TYPED_LOAD), + ); + caps.set( + Tfc::STORAGE_ATOMIC, + data_srv_uav + .Support2 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT2_UAV_ATOMIC_UNSIGNED_MIN_OR_MAX), + ); + caps.set( + Tfc::STORAGE_WRITE_ONLY, + data_srv_uav + .Support2 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT2_UAV_TYPED_STORE), + ); + caps.set( + Tfc::STORAGE_READ_WRITE, + caps.contains(Tfc::STORAGE_READ_ONLY | Tfc::STORAGE_WRITE_ONLY), + ); + + // We load via UAV/SRV so use srv_uav_format + let no_msaa_load = caps.contains(Tfc::SAMPLED) + && !data_srv_uav + .Support1 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT1_MULTISAMPLE_LOAD); + + let no_msaa_target = (data.Support1 + & (Direct3D12::D3D12_FORMAT_SUPPORT1_RENDER_TARGET + | Direct3D12::D3D12_FORMAT_SUPPORT1_DEPTH_STENCIL)) + .0 + != 0 + && !data + .Support1 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT1_MULTISAMPLE_RENDERTARGET); + + caps.set( + Tfc::MULTISAMPLE_RESOLVE, + data.Support1 + .contains(Direct3D12::D3D12_FORMAT_SUPPORT1_MULTISAMPLE_RESOLVE), + ); + + let mut ms_levels = Direct3D12::D3D12_FEATURE_DATA_MULTISAMPLE_QUALITY_LEVELS { + Format: raw_format, + SampleCount: 0, + Flags: Direct3D12::D3D12_MULTISAMPLE_QUALITY_LEVELS_FLAG_NONE, + NumQualityLevels: 0, + }; + + let mut set_sample_count = |sc: u32, tfc: Tfc| { + ms_levels.SampleCount = sc; + + if unsafe { + self.device.CheckFeatureSupport( + Direct3D12::D3D12_FEATURE_MULTISAMPLE_QUALITY_LEVELS, + <*mut _>::cast(&mut ms_levels), + size_of_val(&ms_levels) as u32, + ) + } + .is_ok() + && ms_levels.NumQualityLevels != 0 + { + caps.set(tfc, !no_msaa_load && !no_msaa_target); + } + }; + + set_sample_count(2, Tfc::MULTISAMPLE_X2); + set_sample_count(4, Tfc::MULTISAMPLE_X4); + set_sample_count(8, Tfc::MULTISAMPLE_X8); + set_sample_count(16, Tfc::MULTISAMPLE_X16); + + caps + } + + unsafe fn surface_capabilities( + &self, + surface: &super::Surface, + ) -> Option { + let current_extent = { + match surface.target { + SurfaceTarget::WndHandle(wnd_handle) + | SurfaceTarget::VisualFromWndHandle { + handle: wnd_handle, .. + } => { + let mut rect = Default::default(); + if unsafe { WindowsAndMessaging::GetClientRect(wnd_handle, &mut rect) }.is_ok() + { + Some(wgt::Extent3d { + width: (rect.right - rect.left) as u32, + height: (rect.bottom - rect.top) as u32, + depth_or_array_layers: 1, + }) + } else { + log::warn!("Unable to get the window client rect"); + None + } + } + SurfaceTarget::Visual(_) + | SurfaceTarget::SurfaceHandle(_) + | SurfaceTarget::SwapChainPanel(_) => None, + } + }; + + let mut present_modes = vec![wgt::PresentMode::Mailbox, wgt::PresentMode::Fifo]; + if surface.supports_allow_tearing { + present_modes.push(wgt::PresentMode::Immediate); + } + + Some(crate::SurfaceCapabilities { + formats: vec![ + wgt::TextureFormat::Bgra8UnormSrgb, + wgt::TextureFormat::Bgra8Unorm, + wgt::TextureFormat::Rgba8UnormSrgb, + wgt::TextureFormat::Rgba8Unorm, + wgt::TextureFormat::Rgb10a2Unorm, + wgt::TextureFormat::Rgba16Float, + ], + // See https://learn.microsoft.com/en-us/windows/win32/api/dxgi/nf-dxgi-idxgidevice1-setmaximumframelatency + maximum_frame_latency: 1..=16, + current_extent, + usage: wgt::TextureUses::COLOR_TARGET + | wgt::TextureUses::COPY_SRC + | wgt::TextureUses::COPY_DST, + present_modes, + composite_alpha_modes: match surface.target { + SurfaceTarget::WndHandle(_) => vec![wgt::CompositeAlphaMode::Opaque], + SurfaceTarget::Visual(_) + | SurfaceTarget::VisualFromWndHandle { .. } + | SurfaceTarget::SurfaceHandle(_) + | SurfaceTarget::SwapChainPanel(_) => vec![ + wgt::CompositeAlphaMode::Auto, + wgt::CompositeAlphaMode::Inherit, + wgt::CompositeAlphaMode::Opaque, + wgt::CompositeAlphaMode::PostMultiplied, + wgt::CompositeAlphaMode::PreMultiplied, + ], + }, + }) + } + + unsafe fn get_presentation_timestamp(&self) -> wgt::PresentationTimestamp { + wgt::PresentationTimestamp(self.presentation_timer.get_timestamp_ns()) + } + + fn get_ordered_buffer_usages(&self) -> wgt::BufferUses { + wgt::BufferUses::INCLUSIVE | wgt::BufferUses::MAP_WRITE + } + + // Don't put barriers between inclusive uses + // DX12 implicitly orders renderpasses on the same resources. + fn get_ordered_texture_usages(&self) -> wgt::TextureUses { + wgt::TextureUses::INCLUSIVE + | wgt::TextureUses::COLOR_TARGET + | wgt::TextureUses::DEPTH_STENCIL_WRITE + } +} + +fn get_adapter_pci_info(vendor_id: u32, device_id: u32) -> String { + // SAFETY: SetupDiGetClassDevsW is called with valid parameters + let device_info_set = unsafe { + match SetupDiGetClassDevsW(Some(&GUID_DEVCLASS_DISPLAY), None, None, DIGCF_PRESENT) { + Ok(set) => set, + Err(_) => return String::new(), + } + }; + + struct DeviceInfoSetGuard(HDEVINFO); + impl Drop for DeviceInfoSetGuard { + fn drop(&mut self) { + // SAFETY: device_info_set is a valid HDEVINFO and is only dropped once via this guard + unsafe { + let _ = SetupDiDestroyDeviceInfoList(self.0); + } + } + } + let _guard = DeviceInfoSetGuard(device_info_set); + + let mut device_index = 0u32; + loop { + let mut device_info_data = SP_DEVINFO_DATA { + cbSize: size_of::() as u32, + ..Default::default() + }; + + // SAFETY: device_info_set is a valid HDEVINFO, device_index starts at 0 and + // device_info_data is properly initialized above + unsafe { + if SetupDiEnumDeviceInfo(device_info_set, device_index, &mut device_info_data).is_err() + { + if GetLastError() == ERROR_NO_MORE_ITEMS { + break; + } + device_index += 1; + continue; + } + } + + let mut hardware_id_size = 0u32; + // SAFETY: device_info_set and device_info_data are valid + unsafe { + let _ = SetupDiGetDeviceRegistryPropertyW( + device_info_set, + &device_info_data, + SPDRP_HARDWAREID, + None, + None, + Some(&mut hardware_id_size), + ); + } + + if hardware_id_size == 0 { + device_index += 1; + continue; + } + + let mut hardware_id_buffer = vec![0u8; hardware_id_size as usize]; + // SAFETY: device_info_set and device_info_data are valid + unsafe { + if SetupDiGetDeviceRegistryPropertyW( + device_info_set, + &device_info_data, + SPDRP_HARDWAREID, + None, + Some(&mut hardware_id_buffer), + Some(&mut hardware_id_size), + ) + .is_err() + { + device_index += 1; + continue; + } + } + + let hardware_id_u16: Vec = hardware_id_buffer + .chunks_exact(2) + .map(|chunk| u16::from_le_bytes([chunk[0], chunk[1]])) + .collect(); + let hardware_ids: Vec = hardware_id_u16 + .split(|&c| c == 0) + .filter(|s| !s.is_empty()) + .map(|s| String::from_utf16_lossy(s).to_uppercase()) + .collect(); + + // https://learn.microsoft.com/en-us/windows-hardware/drivers/install/identifiers-for-pci-devices + let expected_id = format!("PCI\\VEN_{vendor_id:04X}&DEV_{device_id:04X}"); + if !hardware_ids.iter().any(|id| id.contains(&expected_id)) { + device_index += 1; + continue; + } + + let mut bus_buffer = [0u8; 4]; + let mut data_size = bus_buffer.len() as u32; + // SAFETY: device_info_set and device_info_data are valid + let bus_number = unsafe { + if SetupDiGetDeviceRegistryPropertyW( + device_info_set, + &device_info_data, + SPDRP_BUSNUMBER, + None, + Some(&mut bus_buffer), + Some(&mut data_size), + ) + .is_err() + { + device_index += 1; + continue; + } + u32::from_le_bytes(bus_buffer) + }; + + let mut addr_buffer = [0u8; 4]; + let mut addr_size = addr_buffer.len() as u32; + // SAFETY: device_info_set and device_info_data are valid + unsafe { + if SetupDiGetDeviceRegistryPropertyW( + device_info_set, + &device_info_data, + SPDRP_ADDRESS, + None, + Some(&mut addr_buffer), + Some(&mut addr_size), + ) + .is_err() + { + device_index += 1; + continue; + } + } + let address = u32::from_le_bytes(addr_buffer); + + // https://learn.microsoft.com/en-us/windows-hardware/drivers/kernel/obtaining-device-configuration-information-at-irql---dispatch-level + let device = (address >> 16) & 0x0000FFFF; + let function = address & 0x0000FFFF; + + // domain:bus:device.function + return format!("{:04x}:{:02x}:{:02x}.{:x}", 0, bus_number, device, function); + } + + String::new() +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/command.rs b/third_party/wgpu-hal-29.0.4/src/dx12/command.rs new file mode 100644 index 0000000..5f7d5ba --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/command.rs @@ -0,0 +1,1859 @@ +use alloc::vec::Vec; +use core::{mem, ops::Range}; + +use windows::{ + core::Interface as _, + Win32::{ + Foundation, + Graphics::{Direct3D12, Dxgi}, + }, +}; + +use super::conv; +use crate::{ + auxil::{ + self, + dxgi::{name::ObjectExt as _, result::HResult as _}, + }, + dx12::borrow_interface_temporarily, + AccelerationStructureEntries, CommandEncoder as _, +}; + +fn make_box(origin: &wgt::Origin3d, size: &crate::CopyExtent) -> Direct3D12::D3D12_BOX { + Direct3D12::D3D12_BOX { + left: origin.x, + top: origin.y, + right: origin.x + size.width, + bottom: origin.y + size.height, + front: origin.z, + back: origin.z + size.depth, + } +} + +impl crate::BufferTextureCopy { + fn to_subresource_footprint( + &self, + format: wgt::TextureFormat, + ) -> Direct3D12::D3D12_PLACED_SUBRESOURCE_FOOTPRINT { + let (block_width, _) = format.block_dimensions(); + Direct3D12::D3D12_PLACED_SUBRESOURCE_FOOTPRINT { + Offset: self.buffer_layout.offset, + Footprint: Direct3D12::D3D12_SUBRESOURCE_FOOTPRINT { + Format: auxil::dxgi::conv::map_texture_format_for_copy( + format, + self.texture_base.aspect, + ) + .unwrap(), + Width: self.size.width, + Height: self.size.height, + Depth: self.size.depth, + RowPitch: { + let actual = self.buffer_layout.bytes_per_row.unwrap_or_else(|| { + // this may happen for single-line updates + let block_size = format + .block_copy_size(Some(self.texture_base.aspect.map())) + .unwrap(); + (self.size.width / block_width) * block_size + }); + wgt::math::align_to(actual, Direct3D12::D3D12_TEXTURE_DATA_PITCH_ALIGNMENT) + }, + }, + } + } +} + +impl super::Temp { + fn prepare_marker(&mut self, marker: &str) -> (&[u16], u32) { + self.marker.clear(); + self.marker.extend(marker.encode_utf16()); + self.marker.push(0); + (&self.marker, self.marker.len() as u32 * 2) + } +} + +impl Drop for super::CommandEncoder { + fn drop(&mut self) { + use crate::CommandEncoder; + unsafe { self.discard_encoding() } + + let mut rtv_pool = self.rtv_pool.lock(); + for handle in self.temp_rtv_handles.drain(..) { + rtv_pool.free_handle(handle); + } + drop(rtv_pool); + + self.counters.command_encoders.sub(1); + } +} + +impl super::CommandEncoder { + unsafe fn begin_pass(&mut self, kind: super::PassKind, label: crate::Label) { + let list = self.list.as_ref().unwrap(); + self.pass.kind = kind; + if let Some(label) = label { + let (wide_label, size) = self.temp.prepare_marker(label); + unsafe { list.BeginEvent(0, Some(wide_label.as_ptr().cast()), size) }; + self.pass.has_label = true; + } + self.pass.dirty_root_elements = 0; + self.pass.dirty_vertex_buffers = 0; + unsafe { + list.SetDescriptorHeaps(&[ + Some(self.shared.heap_views.raw.clone()), + Some(self.shared.sampler_heap.heap().clone()), + ]) + }; + } + + unsafe fn end_pass(&mut self) { + let list = self.list.as_ref().unwrap(); + unsafe { list.SetDescriptorHeaps(&[]) }; + if self.pass.has_label { + unsafe { list.EndEvent() }; + } + self.pass.clear(); + } + + unsafe fn prepare_vertex_buffers(&mut self) { + while self.pass.dirty_vertex_buffers != 0 { + let list = self.list.as_ref().unwrap(); + let index = self.pass.dirty_vertex_buffers.trailing_zeros(); + self.pass.dirty_vertex_buffers ^= 1 << index; + unsafe { + list.IASetVertexBuffers( + index, + Some(&self.pass.vertex_buffers[index as usize..][..1]), + ); + } + } + } + + unsafe fn prepare_draw(&mut self, first_vertex: i32, first_instance: u32) { + unsafe { + self.prepare_vertex_buffers(); + } + if let Some(root_index) = self + .pass + .layout + .special_constants + .as_ref() + .map(|sc| sc.root_index) + { + let needs_update = match self.pass.root_elements[root_index as usize] { + super::RootElement::SpecialConstantBuffer { + first_vertex: other_vertex, + first_instance: other_instance, + other: _, + } => first_vertex != other_vertex || first_instance != other_instance, + _ => true, + }; + if needs_update { + self.pass.dirty_root_elements |= 1 << root_index; + self.pass.root_elements[root_index as usize] = + super::RootElement::SpecialConstantBuffer { + first_vertex, + first_instance, + other: 0, + }; + } + } + self.update_root_elements(); + } + + fn prepare_dispatch(&mut self, count: [u32; 3]) { + if let Some(root_index) = self + .pass + .layout + .special_constants + .as_ref() + .map(|sc| sc.root_index) + { + let needs_update = match self.pass.root_elements[root_index as usize] { + super::RootElement::SpecialConstantBuffer { + first_vertex, + first_instance, + other, + } => [first_vertex as u32, first_instance, other] != count, + _ => true, + }; + if needs_update { + self.pass.dirty_root_elements |= 1 << root_index; + self.pass.root_elements[root_index as usize] = + super::RootElement::SpecialConstantBuffer { + first_vertex: count[0] as i32, + first_instance: count[1], + other: count[2], + }; + } + } + self.update_root_elements(); + } + + // Note: we have to call this lazily before draw calls. Otherwise, D3D complains + // about the root parameters being incompatible with root signature. + fn update_root_elements(&mut self) { + use super::PassKind as Pk; + + while self.pass.dirty_root_elements != 0 { + let list = self.list.as_ref().unwrap(); + let index = self.pass.dirty_root_elements.trailing_zeros(); + self.pass.dirty_root_elements ^= 1 << index; + + match self.pass.root_elements[index as usize] { + super::RootElement::Empty => log::error!("Root index {index} is not bound"), + super::RootElement::Constant => { + let info = self.pass.layout.root_constant_info.as_ref().unwrap(); + + for offset in info.range.clone() { + let val = self.pass.constant_data[offset as usize]; + match self.pass.kind { + Pk::Render => unsafe { + list.SetGraphicsRoot32BitConstant(index, val, offset) + }, + Pk::Compute => unsafe { + list.SetComputeRoot32BitConstant(index, val, offset) + }, + Pk::Transfer => (), + } + } + } + super::RootElement::SpecialConstantBuffer { + first_vertex, + first_instance, + other, + } => match self.pass.kind { + Pk::Render => { + unsafe { list.SetGraphicsRoot32BitConstant(index, first_vertex as u32, 0) }; + unsafe { list.SetGraphicsRoot32BitConstant(index, first_instance, 1) }; + } + Pk::Compute => { + unsafe { list.SetComputeRoot32BitConstant(index, first_vertex as u32, 0) }; + unsafe { list.SetComputeRoot32BitConstant(index, first_instance, 1) }; + unsafe { list.SetComputeRoot32BitConstant(index, other, 2) }; + } + Pk::Transfer => (), + }, + super::RootElement::Table(descriptor) => match self.pass.kind { + Pk::Render => unsafe { list.SetGraphicsRootDescriptorTable(index, descriptor) }, + Pk::Compute => unsafe { list.SetComputeRootDescriptorTable(index, descriptor) }, + Pk::Transfer => (), + }, + super::RootElement::DynamicUniformBuffer { address } => { + let address = address.ptr; + match self.pass.kind { + Pk::Render => unsafe { + list.SetGraphicsRootConstantBufferView(index, address) + }, + Pk::Compute => unsafe { + list.SetComputeRootConstantBufferView(index, address) + }, + Pk::Transfer => (), + } + } + super::RootElement::DynamicOffsetsBuffer { start, end } => { + let values = &self.pass.dynamic_storage_buffer_offsets[start..end]; + + for (offset, &value) in values.iter().enumerate() { + match self.pass.kind { + Pk::Render => unsafe { + list.SetGraphicsRoot32BitConstant(index, value, offset as u32) + }, + Pk::Compute => unsafe { + list.SetComputeRoot32BitConstant(index, value, offset as u32) + }, + Pk::Transfer => (), + } + } + } + super::RootElement::SamplerHeap => match self.pass.kind { + Pk::Render => unsafe { + list.SetGraphicsRootDescriptorTable( + index, + self.shared.sampler_heap.gpu_descriptor_table(), + ) + }, + Pk::Compute => unsafe { + list.SetComputeRootDescriptorTable( + index, + self.shared.sampler_heap.gpu_descriptor_table(), + ) + }, + Pk::Transfer => (), + }, + } + } + } + + fn reset_signature(&mut self, layout: &super::PipelineLayoutShared) { + if let Some(root_index) = layout.special_constants.as_ref().map(|sc| sc.root_index) { + self.pass.root_elements[root_index as usize] = + super::RootElement::SpecialConstantBuffer { + first_vertex: 0, + first_instance: 0, + other: 0, + }; + } + if let Some(root_index) = layout.sampler_heap_root_index { + self.pass.root_elements[root_index as usize] = super::RootElement::SamplerHeap; + } + self.pass.layout = layout.clone(); + self.pass.dirty_root_elements = (1 << layout.total_root_elements) - 1; + } + + fn write_pass_end_timestamp_if_requested(&mut self) { + if let Some((query_set_raw, index)) = self.end_of_pass_timer_query.take() { + use crate::CommandEncoder as _; + unsafe { + self.write_timestamp( + &crate::dx12::QuerySet { + raw: query_set_raw, + raw_ty: Direct3D12::D3D12_QUERY_TYPE_TIMESTAMP, + }, + index, + ); + } + } + } + + unsafe fn buf_tex_intermediate( + &mut self, + region: crate::BufferTextureCopy, + tex_fmt: wgt::TextureFormat, + copy_op: impl FnOnce(&mut Self, &super::Buffer, wgt::BufferSize, crate::BufferTextureCopy) -> T, + ) -> (T, super::Buffer) { + let size = { + let copy_info = region.buffer_layout.get_buffer_texture_copy_info( + tex_fmt, + region.texture_base.aspect.map(), + ®ion.size.into(), + ); + copy_info.unwrap().bytes_in_copy + }; + + let size = wgt::BufferSize::new(size).unwrap(); + + let buffer = { + let (resource, allocation) = + super::suballocation::DeviceAllocationContext::from(&*self) + .create_buffer(&crate::BufferDescriptor { + label: None, + size: size.get(), + usage: wgt::BufferUses::COPY_SRC | wgt::BufferUses::COPY_DST, + memory_flags: crate::MemoryFlags::empty(), + }) + .expect(concat!( + "internal error: ", + "failed to allocate intermediate buffer ", + "for offset alignment" + )); + super::Buffer { + resource, + size: size.get(), + allocation, + } + }; + + let mut region = region; + region.buffer_layout.offset = 0; + + unsafe { + self.transition_buffers( + [crate::BufferBarrier { + buffer: &buffer, + usage: crate::StateTransition { + from: wgt::BufferUses::empty(), + to: wgt::BufferUses::COPY_DST, + }, + }] + .into_iter(), + ) + }; + + let t = copy_op(self, &buffer, size, region); + + unsafe { + self.transition_buffers( + [crate::BufferBarrier { + buffer: &buffer, + usage: crate::StateTransition { + from: wgt::BufferUses::COPY_DST, + to: wgt::BufferUses::COPY_SRC, + }, + }] + .into_iter(), + ) + }; + + (t, buffer) + } +} + +impl crate::CommandEncoder for super::CommandEncoder { + type A = super::Api; + + unsafe fn begin_encoding(&mut self, label: crate::Label) -> Result<(), crate::DeviceError> { + let list = loop { + if let Some(list) = self.free_lists.pop() { + // TODO: Is an error expected here and should we print it? + let reset_result = unsafe { list.Reset(&self.allocator, None) }; + if reset_result.is_ok() { + break Some(list); + } + } else { + break None; + } + }; + + let list = if let Some(list) = list { + list + } else { + unsafe { + self.device.CreateCommandList( + 0, + Direct3D12::D3D12_COMMAND_LIST_TYPE_DIRECT, + &self.allocator, + None, + ) + } + .into_device_result("Create command list")? + }; + + if let Some(label) = label { + list.set_name(label)?; + } + + self.list = Some(list); + self.temp.clear(); + self.pass.clear(); + Ok(()) + } + unsafe fn discard_encoding(&mut self) { + if let Some(list) = self.list.take() { + if unsafe { list.Close() }.is_ok() { + self.free_lists.push(list); + } + } + } + unsafe fn end_encoding(&mut self) -> Result { + let raw = self.list.take().unwrap(); + unsafe { raw.Close() }.into_device_result("GraphicsCommandList::close")?; + Ok(super::CommandBuffer { raw }) + } + unsafe fn reset_all>(&mut self, command_buffers: I) { + self.intermediate_copy_bufs.clear(); + for cmd_buf in command_buffers { + self.free_lists.push(cmd_buf.raw); + } + if let Err(e) = unsafe { self.allocator.Reset() } { + log::error!("ID3D12CommandAllocator::Reset() failed with {e}"); + } + } + + unsafe fn transition_buffers<'a, T>(&mut self, barriers: T) + where + T: Iterator>, + { + self.temp.barriers.clear(); + + for barrier in barriers { + let s0 = conv::map_buffer_usage_to_state(barrier.usage.from); + let s1 = conv::map_buffer_usage_to_state(barrier.usage.to); + if s0 != s1 { + let raw = Direct3D12::D3D12_RESOURCE_BARRIER { + Type: Direct3D12::D3D12_RESOURCE_BARRIER_TYPE_TRANSITION, + Flags: Direct3D12::D3D12_RESOURCE_BARRIER_FLAG_NONE, + Anonymous: Direct3D12::D3D12_RESOURCE_BARRIER_0 { + Transition: mem::ManuallyDrop::new( + Direct3D12::D3D12_RESOURCE_TRANSITION_BARRIER { + pResource: unsafe { + borrow_interface_temporarily(&barrier.buffer.resource) + }, + Subresource: Direct3D12::D3D12_RESOURCE_BARRIER_ALL_SUBRESOURCES, + StateBefore: s0, + StateAfter: s1, + }, + ), + }, + }; + self.temp.barriers.push(raw); + } else if barrier.usage.from == wgt::BufferUses::STORAGE_READ_WRITE + || barrier.usage.from == wgt::BufferUses::ACCELERATION_STRUCTURE_QUERY + { + let raw = Direct3D12::D3D12_RESOURCE_BARRIER { + Type: Direct3D12::D3D12_RESOURCE_BARRIER_TYPE_UAV, + Flags: Direct3D12::D3D12_RESOURCE_BARRIER_FLAG_NONE, + Anonymous: Direct3D12::D3D12_RESOURCE_BARRIER_0 { + UAV: mem::ManuallyDrop::new(Direct3D12::D3D12_RESOURCE_UAV_BARRIER { + pResource: unsafe { + borrow_interface_temporarily(&barrier.buffer.resource) + }, + }), + }, + }; + self.temp.barriers.push(raw); + } + } + + if !self.temp.barriers.is_empty() { + unsafe { + self.list + .as_ref() + .unwrap() + .ResourceBarrier(&self.temp.barriers) + }; + } + } + + unsafe fn transition_textures<'a, T>(&mut self, barriers: T) + where + T: Iterator>, + { + self.temp.barriers.clear(); + + for barrier in barriers { + let s0 = conv::map_texture_usage_to_state(barrier.usage.from); + let s1 = conv::map_texture_usage_to_state(barrier.usage.to); + if s0 != s1 { + let mut raw = Direct3D12::D3D12_RESOURCE_BARRIER { + Type: Direct3D12::D3D12_RESOURCE_BARRIER_TYPE_TRANSITION, + Flags: Direct3D12::D3D12_RESOURCE_BARRIER_FLAG_NONE, + Anonymous: Direct3D12::D3D12_RESOURCE_BARRIER_0 { + Transition: mem::ManuallyDrop::new( + Direct3D12::D3D12_RESOURCE_TRANSITION_BARRIER { + pResource: unsafe { + borrow_interface_temporarily(&barrier.texture.resource) + }, + Subresource: Direct3D12::D3D12_RESOURCE_BARRIER_ALL_SUBRESOURCES, + StateBefore: s0, + StateAfter: s1, + }, + ), + }, + }; + + let tex_mip_level_count = barrier.texture.mip_level_count; + let tex_array_layer_count = barrier.texture.array_layer_count(); + + if barrier.range.is_full_resource( + barrier.texture.format, + tex_mip_level_count, + tex_array_layer_count, + ) { + // Only one barrier if it affects the whole image. + self.temp.barriers.push(raw); + } else { + // Selected texture aspect is relevant if the texture format has both depth _and_ stencil aspects. + let planes = if barrier.texture.format.is_combined_depth_stencil_format() { + match barrier.range.aspect { + wgt::TextureAspect::All => 0..2, + wgt::TextureAspect::DepthOnly => 0..1, + wgt::TextureAspect::StencilOnly => 1..2, + _ => unreachable!(), + } + } else if let Some(planes) = barrier.texture.format.planes() { + match barrier.range.aspect { + wgt::TextureAspect::All => 0..planes, + wgt::TextureAspect::Plane0 => 0..1, + wgt::TextureAspect::Plane1 => 1..2, + wgt::TextureAspect::Plane2 => 2..3, + _ => unreachable!(), + } + } else { + match barrier.texture.format { + wgt::TextureFormat::Stencil8 => 1..2, + wgt::TextureFormat::Depth24Plus => 0..2, // TODO: investigate why tests fail if we set this to 0..1 + _ => 0..1, + } + }; + + for mip_level in barrier.range.mip_range(tex_mip_level_count) { + for array_layer in barrier.range.layer_range(tex_array_layer_count) { + for plane in planes.clone() { + unsafe { &mut *raw.Anonymous.Transition }.Subresource = barrier + .texture + .calc_subresource(mip_level, array_layer, plane); + self.temp.barriers.push(raw.clone()); + } + } + } + } + } else if barrier.usage.from == wgt::TextureUses::STORAGE_READ_WRITE { + let raw = Direct3D12::D3D12_RESOURCE_BARRIER { + Type: Direct3D12::D3D12_RESOURCE_BARRIER_TYPE_UAV, + Flags: Direct3D12::D3D12_RESOURCE_BARRIER_FLAG_NONE, + Anonymous: Direct3D12::D3D12_RESOURCE_BARRIER_0 { + UAV: mem::ManuallyDrop::new(Direct3D12::D3D12_RESOURCE_UAV_BARRIER { + pResource: unsafe { + borrow_interface_temporarily(&barrier.texture.resource) + }, + }), + }, + }; + self.temp.barriers.push(raw); + } + } + + if !self.temp.barriers.is_empty() { + unsafe { + self.list + .as_ref() + .unwrap() + .ResourceBarrier(&self.temp.barriers) + }; + } + } + + unsafe fn clear_buffer(&mut self, buffer: &super::Buffer, range: crate::MemoryRange) { + let list = self.list.as_ref().unwrap(); + let mut offset = range.start; + while offset < range.end { + let size = super::ZERO_BUFFER_SIZE.min(range.end - offset); + unsafe { + list.CopyBufferRegion(&buffer.resource, offset, &self.shared.zero_buffer, 0, size) + }; + offset += size; + } + } + + unsafe fn copy_buffer_to_buffer( + &mut self, + src: &super::Buffer, + dst: &super::Buffer, + regions: T, + ) where + T: Iterator, + { + let list = self.list.as_ref().unwrap(); + for r in regions { + unsafe { + list.CopyBufferRegion( + &dst.resource, + r.dst_offset, + &src.resource, + r.src_offset, + r.size.get(), + ) + }; + } + } + + unsafe fn copy_texture_to_texture( + &mut self, + src: &super::Texture, + _src_usage: wgt::TextureUses, + dst: &super::Texture, + regions: T, + ) where + T: Iterator, + { + let list = self.list.as_ref().unwrap(); + + for r in regions { + let src_location = Direct3D12::D3D12_TEXTURE_COPY_LOCATION { + pResource: unsafe { borrow_interface_temporarily(&src.resource) }, + Type: Direct3D12::D3D12_TEXTURE_COPY_TYPE_SUBRESOURCE_INDEX, + Anonymous: Direct3D12::D3D12_TEXTURE_COPY_LOCATION_0 { + SubresourceIndex: src.calc_subresource_for_copy(&r.src_base), + }, + }; + let dst_location = Direct3D12::D3D12_TEXTURE_COPY_LOCATION { + pResource: unsafe { borrow_interface_temporarily(&dst.resource) }, + Type: Direct3D12::D3D12_TEXTURE_COPY_TYPE_SUBRESOURCE_INDEX, + Anonymous: Direct3D12::D3D12_TEXTURE_COPY_LOCATION_0 { + SubresourceIndex: dst.calc_subresource_for_copy(&r.dst_base), + }, + }; + + let src_box = make_box(&r.src_base.origin, &r.size); + + unsafe { + list.CopyTextureRegion( + &dst_location, + r.dst_base.origin.x, + r.dst_base.origin.y, + r.dst_base.origin.z, + &src_location, + Some(&src_box), + ) + }; + } + } + + unsafe fn copy_buffer_to_texture( + &mut self, + src: &super::Buffer, + dst: &super::Texture, + regions: T, + ) where + T: Iterator, + { + let offset_alignment = self.shared.private_caps.texture_data_placement_alignment(); + + for naive_copy_region in regions { + let is_offset_aligned = naive_copy_region.buffer_layout.offset % offset_alignment == 0; + let (final_copy_region, src) = if is_offset_aligned { + (naive_copy_region, src) + } else { + let (intermediate_to_dst_region, intermediate_buf) = unsafe { + let src_offset = naive_copy_region.buffer_layout.offset; + self.buf_tex_intermediate( + naive_copy_region, + dst.format, + |this, buf, size, intermediate_to_dst_region| { + let layout = crate::BufferCopy { + src_offset, + dst_offset: 0, + size, + }; + this.copy_buffer_to_buffer(src, buf, [layout].into_iter()); + intermediate_to_dst_region + }, + ) + }; + self.intermediate_copy_bufs.push(intermediate_buf); + let intermediate_buf = self.intermediate_copy_bufs.last().unwrap(); + (intermediate_to_dst_region, intermediate_buf) + }; + + let list = self.list.as_ref().unwrap(); + + let src_location = Direct3D12::D3D12_TEXTURE_COPY_LOCATION { + pResource: unsafe { borrow_interface_temporarily(&src.resource) }, + Type: Direct3D12::D3D12_TEXTURE_COPY_TYPE_PLACED_FOOTPRINT, + Anonymous: Direct3D12::D3D12_TEXTURE_COPY_LOCATION_0 { + PlacedFootprint: final_copy_region.to_subresource_footprint(dst.format), + }, + }; + let dst_location = Direct3D12::D3D12_TEXTURE_COPY_LOCATION { + pResource: unsafe { borrow_interface_temporarily(&dst.resource) }, + Type: Direct3D12::D3D12_TEXTURE_COPY_TYPE_SUBRESOURCE_INDEX, + Anonymous: Direct3D12::D3D12_TEXTURE_COPY_LOCATION_0 { + SubresourceIndex: dst + .calc_subresource_for_copy(&final_copy_region.texture_base), + }, + }; + + let src_box = make_box(&wgt::Origin3d::ZERO, &final_copy_region.size); + unsafe { + list.CopyTextureRegion( + &dst_location, + final_copy_region.texture_base.origin.x, + final_copy_region.texture_base.origin.y, + final_copy_region.texture_base.origin.z, + &src_location, + Some(&src_box), + ) + }; + } + } + + unsafe fn copy_texture_to_buffer( + &mut self, + src: &super::Texture, + _src_usage: wgt::TextureUses, + dst: &super::Buffer, + regions: T, + ) where + T: Iterator, + { + let copy_aligned = |this: &mut Self, + src: &super::Texture, + dst: &super::Buffer, + r: crate::BufferTextureCopy| { + let list = this.list.as_ref().unwrap(); + + let src_location = Direct3D12::D3D12_TEXTURE_COPY_LOCATION { + pResource: unsafe { borrow_interface_temporarily(&src.resource) }, + Type: Direct3D12::D3D12_TEXTURE_COPY_TYPE_SUBRESOURCE_INDEX, + Anonymous: Direct3D12::D3D12_TEXTURE_COPY_LOCATION_0 { + SubresourceIndex: src.calc_subresource_for_copy(&r.texture_base), + }, + }; + let dst_location = Direct3D12::D3D12_TEXTURE_COPY_LOCATION { + pResource: unsafe { borrow_interface_temporarily(&dst.resource) }, + Type: Direct3D12::D3D12_TEXTURE_COPY_TYPE_PLACED_FOOTPRINT, + Anonymous: Direct3D12::D3D12_TEXTURE_COPY_LOCATION_0 { + PlacedFootprint: r.to_subresource_footprint(src.format), + }, + }; + + let src_box = make_box(&r.texture_base.origin, &r.size); + unsafe { + list.CopyTextureRegion(&dst_location, 0, 0, 0, &src_location, Some(&src_box)) + }; + }; + + let offset_alignment = self.shared.private_caps.texture_data_placement_alignment(); + + for r in regions { + let is_offset_aligned = r.buffer_layout.offset % offset_alignment == 0; + if is_offset_aligned { + copy_aligned(self, src, dst, r) + } else { + let orig_offset = r.buffer_layout.offset; + let (intermediate_to_dst_region, src) = unsafe { + self.buf_tex_intermediate( + r, + src.format, + |this, buf, size, intermediate_region| { + copy_aligned(this, src, buf, intermediate_region); + crate::BufferCopy { + src_offset: 0, + dst_offset: orig_offset, + size, + } + }, + ) + }; + + unsafe { + self.copy_buffer_to_buffer(&src, dst, [intermediate_to_dst_region].into_iter()); + } + + self.intermediate_copy_bufs.push(src); + }; + } + } + + unsafe fn begin_query(&mut self, set: &super::QuerySet, index: u32) { + unsafe { + self.list + .as_ref() + .unwrap() + .BeginQuery(&set.raw, set.raw_ty, index) + }; + } + unsafe fn end_query(&mut self, set: &super::QuerySet, index: u32) { + unsafe { + self.list + .as_ref() + .unwrap() + .EndQuery(&set.raw, set.raw_ty, index) + }; + } + unsafe fn write_timestamp(&mut self, set: &super::QuerySet, index: u32) { + unsafe { + self.list.as_ref().unwrap().EndQuery( + &set.raw, + Direct3D12::D3D12_QUERY_TYPE_TIMESTAMP, + index, + ) + }; + } + unsafe fn read_acceleration_structure_compact_size( + &mut self, + acceleration_structure: &super::AccelerationStructure, + buf: &super::Buffer, + ) { + let list = self + .list + .as_ref() + .unwrap() + .cast::() + .unwrap(); + unsafe { + list.EmitRaytracingAccelerationStructurePostbuildInfo( + &Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_POSTBUILD_INFO_DESC { + DestBuffer: buf.resource.GetGPUVirtualAddress(), + InfoType: Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_POSTBUILD_INFO_COMPACTED_SIZE, + }, + &[ + acceleration_structure.resource.GetGPUVirtualAddress() + ], + ) + } + } + unsafe fn reset_queries(&mut self, _set: &super::QuerySet, _range: Range) { + // nothing to do here + } + unsafe fn copy_query_results( + &mut self, + set: &super::QuerySet, + range: Range, + buffer: &super::Buffer, + offset: wgt::BufferAddress, + _stride: wgt::BufferSize, + ) { + unsafe { + self.list.as_ref().unwrap().ResolveQueryData( + &set.raw, + set.raw_ty, + range.start, + range.end - range.start, + &buffer.resource, + offset, + ) + }; + } + + // render + + unsafe fn begin_render_pass( + &mut self, + desc: &crate::RenderPassDescriptor, + ) -> Result<(), crate::DeviceError> { + unsafe { self.begin_pass(super::PassKind::Render, desc.label) }; + + // Start timestamp if any (before all other commands but after debug marker) + if let Some(timestamp_writes) = desc.timestamp_writes.as_ref() { + if let Some(index) = timestamp_writes.beginning_of_pass_write_index { + unsafe { + self.write_timestamp(timestamp_writes.query_set, index); + } + } + self.end_of_pass_timer_query = timestamp_writes + .end_of_pass_write_index + .map(|index| (timestamp_writes.query_set.raw.clone(), index)); + } + + let mut color_views = + [Direct3D12::D3D12_CPU_DESCRIPTOR_HANDLE { ptr: 0 }; crate::MAX_COLOR_ATTACHMENTS]; + let mut rtv_pool = self.rtv_pool.lock(); + for (rtv, cat) in color_views.iter_mut().zip(desc.color_attachments.iter()) { + if let Some(cat) = cat.as_ref() { + if cat.target.view.dimension == wgt::TextureViewDimension::D3 { + let desc = Direct3D12::D3D12_RENDER_TARGET_VIEW_DESC { + Format: cat.target.view.raw_format, + ViewDimension: Direct3D12::D3D12_RTV_DIMENSION_TEXTURE3D, + Anonymous: Direct3D12::D3D12_RENDER_TARGET_VIEW_DESC_0 { + Texture3D: Direct3D12::D3D12_TEX3D_RTV { + MipSlice: cat.target.view.mip_slice, + FirstWSlice: cat.depth_slice.unwrap(), + WSize: 1, + }, + }, + }; + let handle = rtv_pool.alloc_handle()?; + unsafe { + self.device.CreateRenderTargetView( + &cat.target.view.texture, + Some(&desc), + handle.raw, + ) + }; + *rtv = handle.raw; + self.temp_rtv_handles.push(handle); + } else { + *rtv = cat.target.view.handle_rtv.unwrap().raw; + } + } else { + *rtv = self.null_rtv_handle.raw; + } + } + drop(rtv_pool); + + let ds_view = desc.depth_stencil_attachment.as_ref().map(|ds| { + if ds.target.usage == wgt::TextureUses::DEPTH_STENCIL_WRITE { + ds.target.view.handle_dsv_rw.as_ref().unwrap().raw + } else { + ds.target.view.handle_dsv_ro.as_ref().unwrap().raw + } + }); + + let list = self.list.as_ref().unwrap(); + unsafe { + list.OMSetRenderTargets( + desc.color_attachments.len() as u32, + Some(color_views.as_ptr()), + false, + ds_view.as_ref().map(core::ptr::from_ref), + ) + }; + + self.pass.resolves.clear(); + for (rtv, cat) in color_views.iter().zip(desc.color_attachments.iter()) { + if let Some(cat) = cat.as_ref() { + if cat.ops.contains(crate::AttachmentOps::LOAD_CLEAR) { + let value = [ + cat.clear_value.r as f32, + cat.clear_value.g as f32, + cat.clear_value.b as f32, + cat.clear_value.a as f32, + ]; + unsafe { list.ClearRenderTargetView(*rtv, &value, None) }; + } + if let Some(ref target) = cat.resolve_target { + self.pass.resolves.push(super::PassResolve { + src: ( + cat.target.view.texture.clone(), + cat.target.view.subresource_index, + ), + dst: (target.view.texture.clone(), target.view.subresource_index), + format: target.view.raw_format, + }); + } + } + } + + if let Some(ref ds) = desc.depth_stencil_attachment { + let mut flags = Direct3D12::D3D12_CLEAR_FLAGS::default(); + let aspects = ds.target.view.aspects; + if ds.depth_ops.contains(crate::AttachmentOps::LOAD_CLEAR) + && aspects.contains(crate::FormatAspects::DEPTH) + { + flags |= Direct3D12::D3D12_CLEAR_FLAG_DEPTH; + } + if ds.stencil_ops.contains(crate::AttachmentOps::LOAD_CLEAR) + && aspects.contains(crate::FormatAspects::STENCIL) + { + flags |= Direct3D12::D3D12_CLEAR_FLAG_STENCIL; + } + + if let Some(ds_view) = ds_view { + if flags != Direct3D12::D3D12_CLEAR_FLAGS::default() { + unsafe { + list.ClearDepthStencilView( + ds_view, + flags, + ds.clear_value.0, + ds.clear_value.1 as u8, + None, + ) + } + } + } + } + + if let Some(multiview_mask) = desc.multiview_mask { + unsafe { + list.cast::() + .unwrap() + .SetViewInstanceMask(multiview_mask.get()); + } + } + + let raw_vp = Direct3D12::D3D12_VIEWPORT { + TopLeftX: 0.0, + TopLeftY: 0.0, + Width: desc.extent.width as f32, + Height: desc.extent.height as f32, + MinDepth: 0.0, + MaxDepth: 1.0, + }; + let raw_rect = Foundation::RECT { + left: 0, + top: 0, + right: desc.extent.width as i32, + bottom: desc.extent.height as i32, + }; + unsafe { list.RSSetViewports(core::slice::from_ref(&raw_vp)) }; + unsafe { list.RSSetScissorRects(core::slice::from_ref(&raw_rect)) }; + + Ok(()) + } + + unsafe fn end_render_pass(&mut self) { + if !self.pass.resolves.is_empty() { + let list = self.list.as_ref().unwrap(); + self.temp.barriers.clear(); + + // All the targets are expected to be in `COLOR_TARGET` state, + // but D3D12 has special source/destination states for the resolves. + for resolve in self.pass.resolves.iter() { + let barrier = Direct3D12::D3D12_RESOURCE_BARRIER { + Type: Direct3D12::D3D12_RESOURCE_BARRIER_TYPE_TRANSITION, + Flags: Direct3D12::D3D12_RESOURCE_BARRIER_FLAG_NONE, + Anonymous: Direct3D12::D3D12_RESOURCE_BARRIER_0 { + // Note: this assumes `D3D12_RESOURCE_STATE_RENDER_TARGET`. + // If it's not the case, we can include the `TextureUses` in `PassResolve`. + Transition: mem::ManuallyDrop::new( + Direct3D12::D3D12_RESOURCE_TRANSITION_BARRIER { + pResource: unsafe { borrow_interface_temporarily(&resolve.src.0) }, + Subresource: resolve.src.1, + StateBefore: Direct3D12::D3D12_RESOURCE_STATE_RENDER_TARGET, + StateAfter: Direct3D12::D3D12_RESOURCE_STATE_RESOLVE_SOURCE, + }, + ), + }, + }; + self.temp.barriers.push(barrier); + let barrier = Direct3D12::D3D12_RESOURCE_BARRIER { + Type: Direct3D12::D3D12_RESOURCE_BARRIER_TYPE_TRANSITION, + Flags: Direct3D12::D3D12_RESOURCE_BARRIER_FLAG_NONE, + Anonymous: Direct3D12::D3D12_RESOURCE_BARRIER_0 { + // Note: this assumes `D3D12_RESOURCE_STATE_RENDER_TARGET`. + // If it's not the case, we can include the `TextureUses` in `PassResolve`. + Transition: mem::ManuallyDrop::new( + Direct3D12::D3D12_RESOURCE_TRANSITION_BARRIER { + pResource: unsafe { borrow_interface_temporarily(&resolve.dst.0) }, + Subresource: resolve.dst.1, + StateBefore: Direct3D12::D3D12_RESOURCE_STATE_RENDER_TARGET, + StateAfter: Direct3D12::D3D12_RESOURCE_STATE_RESOLVE_DEST, + }, + ), + }, + }; + self.temp.barriers.push(barrier); + } + + if !self.temp.barriers.is_empty() { + profiling::scope!("ID3D12GraphicsCommandList::ResourceBarrier"); + unsafe { list.ResourceBarrier(&self.temp.barriers) }; + } + + for resolve in self.pass.resolves.iter() { + profiling::scope!("ID3D12GraphicsCommandList::ResolveSubresource"); + unsafe { + list.ResolveSubresource( + &resolve.dst.0, + resolve.dst.1, + &resolve.src.0, + resolve.src.1, + resolve.format, + ) + }; + } + + // Flip all the barriers to reverse, back into `COLOR_TARGET`. + for barrier in self.temp.barriers.iter_mut() { + let transition = unsafe { &mut *barrier.Anonymous.Transition }; + mem::swap(&mut transition.StateBefore, &mut transition.StateAfter); + } + if !self.temp.barriers.is_empty() { + profiling::scope!("ID3D12GraphicsCommandList::ResourceBarrier"); + unsafe { list.ResourceBarrier(&self.temp.barriers) }; + } + } + + self.write_pass_end_timestamp_if_requested(); + + unsafe { self.end_pass() }; + } + + unsafe fn set_bind_group( + &mut self, + layout: &super::PipelineLayout, + index: u32, + group: &super::BindGroup, + dynamic_offsets: &[wgt::DynamicOffset], + ) { + let info = layout.bind_group_infos[index as usize].as_ref().unwrap(); + let mut root_index = info.base_root_index as usize; + + // Bind CBV/SRC/UAV descriptor tables + if info.tables.contains(super::TableTypes::SRV_CBV_UAV) { + self.pass.root_elements[root_index] = + super::RootElement::Table(group.handle_views.unwrap().gpu); + root_index += 1; + } + + let mut offsets_index = 0; + if let Some(dynamic_storage_buffer_offsets) = info.dynamic_storage_buffer_offsets.as_ref() { + let root_index = dynamic_storage_buffer_offsets.root_index; + let range = &dynamic_storage_buffer_offsets.range; + + if range.end > self.pass.dynamic_storage_buffer_offsets.len() { + self.pass + .dynamic_storage_buffer_offsets + .resize(range.end, 0); + } + + offsets_index += range.start; + + self.pass.root_elements[root_index as usize] = + super::RootElement::DynamicOffsetsBuffer { + start: range.start, + end: range.end, + }; + + if self.pass.layout.signature == layout.shared.signature { + self.pass.dirty_root_elements |= 1 << root_index; + } else { + // D3D12 requires full reset on signature change + // but we don't reset it here since it will be reset below + }; + } + + // Bind root descriptors for dynamic uniform buffers + // or set root constants for offsets of dynamic storage buffers + for (&dynamic_buffer, &offset) in group.dynamic_buffers.iter().zip(dynamic_offsets) { + match dynamic_buffer { + super::DynamicBuffer::Uniform(gpu_base) => { + self.pass.root_elements[root_index] = + super::RootElement::DynamicUniformBuffer { + address: Direct3D12::D3D12_GPU_DESCRIPTOR_HANDLE { + ptr: gpu_base.ptr + offset as u64, + }, + }; + root_index += 1; + } + super::DynamicBuffer::Storage => { + self.pass.dynamic_storage_buffer_offsets[offsets_index] = offset; + offsets_index += 1; + } + } + } + + if self.pass.layout.signature == layout.shared.signature { + self.pass.dirty_root_elements |= (1 << root_index) - (1 << info.base_root_index); + } else { + // D3D12 requires full reset on signature change + self.reset_signature(&layout.shared); + }; + } + unsafe fn set_immediates( + &mut self, + layout: &super::PipelineLayout, + offset_bytes: u32, + data: &[u32], + ) { + let offset_words = offset_bytes as usize / 4; + + let info = layout.shared.root_constant_info.as_ref().unwrap(); + + self.pass.root_elements[info.root_index as usize] = super::RootElement::Constant; + + self.pass.constant_data[offset_words..(offset_words + data.len())].copy_from_slice(data); + + if self.pass.layout.signature == layout.shared.signature { + self.pass.dirty_root_elements |= 1 << info.root_index; + } else { + // D3D12 requires full reset on signature change + self.reset_signature(&layout.shared); + }; + } + + unsafe fn insert_debug_marker(&mut self, label: &str) { + let (wide_label, size) = self.temp.prepare_marker(label); + unsafe { + self.list + .as_ref() + .unwrap() + .SetMarker(0, Some(wide_label.as_ptr().cast()), size) + }; + } + unsafe fn begin_debug_marker(&mut self, group_label: &str) { + let (wide_label, size) = self.temp.prepare_marker(group_label); + unsafe { + self.list + .as_ref() + .unwrap() + .BeginEvent(0, Some(wide_label.as_ptr().cast()), size) + }; + } + unsafe fn end_debug_marker(&mut self) { + unsafe { self.list.as_ref().unwrap().EndEvent() } + } + + unsafe fn set_render_pipeline(&mut self, pipeline: &super::RenderPipeline) { + let list = self.list.clone().unwrap(); + + if self.pass.layout.signature != pipeline.layout.signature { + // D3D12 requires full reset on signature change + unsafe { list.SetGraphicsRootSignature(pipeline.layout.signature.as_ref()) }; + self.reset_signature(&pipeline.layout); + }; + + unsafe { list.SetPipelineState(&pipeline.raw) }; + unsafe { list.IASetPrimitiveTopology(pipeline.topology) }; + + for (index, (vb, &stride)) in self + .pass + .vertex_buffers + .iter_mut() + .zip(pipeline.vertex_strides.iter()) + .enumerate() + { + if let Some(stride) = stride { + if vb.StrideInBytes != stride { + vb.StrideInBytes = stride; + self.pass.dirty_vertex_buffers |= 1 << index; + } + } + } + } + + unsafe fn set_index_buffer<'a>( + &mut self, + binding: crate::BufferBinding<'a, super::Buffer>, + format: wgt::IndexFormat, + ) { + let ibv = Direct3D12::D3D12_INDEX_BUFFER_VIEW { + BufferLocation: binding.resolve_address(), + SizeInBytes: binding.resolve_size().try_into().unwrap(), + Format: auxil::dxgi::conv::map_index_format(format), + }; + + unsafe { self.list.as_ref().unwrap().IASetIndexBuffer(Some(&ibv)) } + } + unsafe fn set_vertex_buffer<'a>( + &mut self, + index: u32, + binding: crate::BufferBinding<'a, super::Buffer>, + ) { + let vb = &mut self.pass.vertex_buffers[index as usize]; + vb.BufferLocation = binding.resolve_address(); + vb.SizeInBytes = binding.resolve_size().try_into().unwrap(); + self.pass.dirty_vertex_buffers |= 1 << index; + } + + unsafe fn set_viewport(&mut self, rect: &crate::Rect, depth_range: Range) { + let raw_vp = Direct3D12::D3D12_VIEWPORT { + TopLeftX: rect.x, + TopLeftY: rect.y, + Width: rect.w, + Height: rect.h, + MinDepth: depth_range.start, + MaxDepth: depth_range.end, + }; + unsafe { + self.list + .as_ref() + .unwrap() + .RSSetViewports(core::slice::from_ref(&raw_vp)) + } + } + unsafe fn set_scissor_rect(&mut self, rect: &crate::Rect) { + let raw_rect = Foundation::RECT { + left: rect.x as i32, + top: rect.y as i32, + right: (rect.x + rect.w) as i32, + bottom: (rect.y + rect.h) as i32, + }; + unsafe { + self.list + .as_ref() + .unwrap() + .RSSetScissorRects(core::slice::from_ref(&raw_rect)) + } + } + unsafe fn set_stencil_reference(&mut self, value: u32) { + unsafe { self.list.as_ref().unwrap().OMSetStencilRef(value) } + } + unsafe fn set_blend_constants(&mut self, color: &[f32; 4]) { + unsafe { self.list.as_ref().unwrap().OMSetBlendFactor(Some(color)) } + } + + unsafe fn draw( + &mut self, + first_vertex: u32, + vertex_count: u32, + first_instance: u32, + instance_count: u32, + ) { + unsafe { self.prepare_draw(first_vertex as i32, first_instance) }; + unsafe { + self.list.as_ref().unwrap().DrawInstanced( + vertex_count, + instance_count, + first_vertex, + first_instance, + ) + } + } + unsafe fn draw_indexed( + &mut self, + first_index: u32, + index_count: u32, + base_vertex: i32, + first_instance: u32, + instance_count: u32, + ) { + unsafe { self.prepare_draw(base_vertex, first_instance) }; + unsafe { + self.list.as_ref().unwrap().DrawIndexedInstanced( + index_count, + instance_count, + first_index, + base_vertex, + first_instance, + ) + } + } + unsafe fn draw_mesh_tasks( + &mut self, + group_count_x: u32, + group_count_y: u32, + group_count_z: u32, + ) { + self.prepare_dispatch([group_count_x, group_count_y, group_count_z]); + let cmd_list6: Direct3D12::ID3D12GraphicsCommandList6 = + self.list.as_ref().unwrap().cast().unwrap(); + unsafe { + cmd_list6.DispatchMesh(group_count_x, group_count_y, group_count_z); + } + } + unsafe fn draw_indirect( + &mut self, + buffer: &super::Buffer, + offset: wgt::BufferAddress, + draw_count: u32, + ) { + if self + .pass + .layout + .special_constants + .as_ref() + .and_then(|sc| sc.indirect_cmd_signatures.as_ref()) + .is_some() + { + unsafe { self.prepare_vertex_buffers() }; + self.update_root_elements(); + } else { + unsafe { self.prepare_draw(0, 0) }; + } + + let cmd_signature = &self + .pass + .layout + .special_constants + .as_ref() + .and_then(|sc| sc.indirect_cmd_signatures.as_ref()) + .unwrap_or_else(|| &self.shared.cmd_signatures) + .draw; + unsafe { + self.list.as_ref().unwrap().ExecuteIndirect( + cmd_signature, + draw_count, + &buffer.resource, + offset, + None, + 0, + ) + } + } + unsafe fn draw_indexed_indirect( + &mut self, + buffer: &super::Buffer, + offset: wgt::BufferAddress, + draw_count: u32, + ) { + if self + .pass + .layout + .special_constants + .as_ref() + .and_then(|sc| sc.indirect_cmd_signatures.as_ref()) + .is_some() + { + unsafe { self.prepare_vertex_buffers() }; + self.update_root_elements(); + } else { + unsafe { self.prepare_draw(0, 0) }; + } + + let cmd_signature = &self + .pass + .layout + .special_constants + .as_ref() + .and_then(|sc| sc.indirect_cmd_signatures.as_ref()) + .unwrap_or_else(|| &self.shared.cmd_signatures) + .draw_indexed; + unsafe { + self.list.as_ref().unwrap().ExecuteIndirect( + cmd_signature, + draw_count, + &buffer.resource, + offset, + None, + 0, + ) + } + } + unsafe fn draw_mesh_tasks_indirect( + &mut self, + buffer: &::Buffer, + offset: wgt::BufferAddress, + draw_count: u32, + ) { + if self + .pass + .layout + .special_constants + .as_ref() + .and_then(|sc| sc.indirect_cmd_signatures.as_ref()) + .is_some() + { + self.update_root_elements(); + } else { + self.prepare_dispatch([0; 3]); + } + + let cmd_list6: Direct3D12::ID3D12GraphicsCommandList6 = + self.list.as_ref().unwrap().cast().unwrap(); + let Some(cmd_signature) = &self + .pass + .layout + .special_constants + .as_ref() + .and_then(|sc| sc.indirect_cmd_signatures.as_ref()) + .unwrap_or_else(|| &self.shared.cmd_signatures) + .draw_mesh + else { + panic!("Feature `MESH_SHADING` not enabled"); + }; + unsafe { + cmd_list6.ExecuteIndirect(cmd_signature, draw_count, &buffer.resource, offset, None, 0); + } + } + unsafe fn draw_indirect_count( + &mut self, + buffer: &super::Buffer, + offset: wgt::BufferAddress, + count_buffer: &super::Buffer, + count_offset: wgt::BufferAddress, + max_count: u32, + ) { + unsafe { self.prepare_draw(0, 0) }; + unsafe { + self.list.as_ref().unwrap().ExecuteIndirect( + &self.shared.cmd_signatures.draw, + max_count, + &buffer.resource, + offset, + &count_buffer.resource, + count_offset, + ) + } + } + unsafe fn draw_indexed_indirect_count( + &mut self, + buffer: &super::Buffer, + offset: wgt::BufferAddress, + count_buffer: &super::Buffer, + count_offset: wgt::BufferAddress, + max_count: u32, + ) { + unsafe { self.prepare_draw(0, 0) }; + unsafe { + self.list.as_ref().unwrap().ExecuteIndirect( + &self.shared.cmd_signatures.draw_indexed, + max_count, + &buffer.resource, + offset, + &count_buffer.resource, + count_offset, + ) + } + } + unsafe fn draw_mesh_tasks_indirect_count( + &mut self, + buffer: &::Buffer, + offset: wgt::BufferAddress, + count_buffer: &::Buffer, + count_offset: wgt::BufferAddress, + max_count: u32, + ) { + self.prepare_dispatch([0; 3]); + let cmd_list6: Direct3D12::ID3D12GraphicsCommandList6 = + self.list.as_ref().unwrap().cast().unwrap(); + let Some(ref command_signature) = self.shared.cmd_signatures.draw_mesh else { + panic!("Feature `MESH_SHADING` not enabled"); + }; + unsafe { + cmd_list6.ExecuteIndirect( + command_signature, + max_count, + &buffer.resource, + offset, + &count_buffer.resource, + count_offset, + ); + } + } + + // compute + + unsafe fn begin_compute_pass<'a>( + &mut self, + desc: &crate::ComputePassDescriptor<'a, super::QuerySet>, + ) { + unsafe { self.begin_pass(super::PassKind::Compute, desc.label) }; + + if let Some(timestamp_writes) = desc.timestamp_writes.as_ref() { + if let Some(index) = timestamp_writes.beginning_of_pass_write_index { + unsafe { + self.write_timestamp(timestamp_writes.query_set, index); + } + } + self.end_of_pass_timer_query = timestamp_writes + .end_of_pass_write_index + .map(|index| (timestamp_writes.query_set.raw.clone(), index)); + } + } + unsafe fn end_compute_pass(&mut self) { + self.write_pass_end_timestamp_if_requested(); + unsafe { self.end_pass() }; + } + + unsafe fn set_compute_pipeline(&mut self, pipeline: &super::ComputePipeline) { + let list = self.list.clone().unwrap(); + + if self.pass.layout.signature != pipeline.layout.signature { + // D3D12 requires full reset on signature change + unsafe { list.SetComputeRootSignature(pipeline.layout.signature.as_ref()) }; + self.reset_signature(&pipeline.layout); + }; + + unsafe { list.SetPipelineState(&pipeline.raw) } + } + + unsafe fn dispatch(&mut self, count @ [x, y, z]: [u32; 3]) { + self.prepare_dispatch(count); + unsafe { self.list.as_ref().unwrap().Dispatch(x, y, z) } + } + + unsafe fn dispatch_indirect(&mut self, buffer: &super::Buffer, offset: wgt::BufferAddress) { + if self + .pass + .layout + .special_constants + .as_ref() + .and_then(|sc| sc.indirect_cmd_signatures.as_ref()) + .is_some() + { + self.update_root_elements(); + } else { + self.prepare_dispatch([0; 3]); + } + + let cmd_signature = &self + .pass + .layout + .special_constants + .as_ref() + .and_then(|sc| sc.indirect_cmd_signatures.as_ref()) + .unwrap_or_else(|| &self.shared.cmd_signatures) + .dispatch; + unsafe { + self.list.as_ref().unwrap().ExecuteIndirect( + cmd_signature, + 1, + &buffer.resource, + offset, + None, + 0, + ) + } + } + + unsafe fn build_acceleration_structures<'a, T>( + &mut self, + _descriptor_count: u32, + descriptors: T, + ) where + super::Api: 'a, + T: IntoIterator< + Item = crate::BuildAccelerationStructureDescriptor< + 'a, + super::Buffer, + super::AccelerationStructure, + >, + >, + { + // Implement using `BuildRaytracingAccelerationStructure`: + // https://microsoft.github.io/DirectX-Specs/d3d/Raytracing.html#buildraytracingaccelerationstructure + let list = self + .list + .as_ref() + .unwrap() + .cast::() + .unwrap(); + for descriptor in descriptors { + // TODO: This is the same as getting build sizes apart from requiring buffers, should this be de-duped? + let mut geometry_desc; + let ty; + let inputs0; + let num_desc; + match descriptor.entries { + AccelerationStructureEntries::Instances(instances) => { + let desc_address = unsafe { + instances + .buffer + .expect("needs buffer to build") + .resource + .GetGPUVirtualAddress() + } + instances.offset as u64; + ty = Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_TYPE_TOP_LEVEL; + inputs0 = Direct3D12::D3D12_BUILD_RAYTRACING_ACCELERATION_STRUCTURE_INPUTS_0 { + InstanceDescs: desc_address, + }; + num_desc = instances.count; + } + AccelerationStructureEntries::Triangles(triangles) => { + geometry_desc = Vec::with_capacity(triangles.len()); + for triangle in triangles { + let transform_address = + triangle.transform.as_ref().map_or(0, |transform| unsafe { + transform.buffer.resource.GetGPUVirtualAddress() + + transform.offset as u64 + }); + let index_format = triangle + .indices + .as_ref() + .map_or(Dxgi::Common::DXGI_FORMAT_UNKNOWN, |indices| { + auxil::dxgi::conv::map_index_format(indices.format) + }); + let vertex_format = + auxil::dxgi::conv::map_vertex_format(triangle.vertex_format); + let index_count = + triangle.indices.as_ref().map_or(0, |indices| indices.count); + let index_address = triangle.indices.as_ref().map_or(0, |indices| unsafe { + indices + .buffer + .expect("needs buffer to build") + .resource + .GetGPUVirtualAddress() + + indices.offset as u64 + }); + let vertex_address = unsafe { + triangle + .vertex_buffer + .expect("needs buffer to build") + .resource + .GetGPUVirtualAddress() + + (triangle.first_vertex as u64 * triangle.vertex_stride) + }; + + let triangle_desc = Direct3D12::D3D12_RAYTRACING_GEOMETRY_TRIANGLES_DESC { + Transform3x4: transform_address, + IndexFormat: index_format, + VertexFormat: vertex_format, + IndexCount: index_count, + VertexCount: triangle.vertex_count, + IndexBuffer: index_address, + VertexBuffer: Direct3D12::D3D12_GPU_VIRTUAL_ADDRESS_AND_STRIDE { + StartAddress: vertex_address, + StrideInBytes: triangle.vertex_stride, + }, + }; + + geometry_desc.push(Direct3D12::D3D12_RAYTRACING_GEOMETRY_DESC { + Type: Direct3D12::D3D12_RAYTRACING_GEOMETRY_TYPE_TRIANGLES, + Flags: conv::map_acceleration_structure_geometry_flags(triangle.flags), + Anonymous: Direct3D12::D3D12_RAYTRACING_GEOMETRY_DESC_0 { + Triangles: triangle_desc, + }, + }) + } + ty = Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_TYPE_BOTTOM_LEVEL; + inputs0 = Direct3D12::D3D12_BUILD_RAYTRACING_ACCELERATION_STRUCTURE_INPUTS_0 { + pGeometryDescs: geometry_desc.as_ptr(), + }; + num_desc = geometry_desc.len() as u32; + } + AccelerationStructureEntries::AABBs(aabbs) => { + geometry_desc = Vec::with_capacity(aabbs.len()); + for aabb in aabbs { + let aabb_address = unsafe { + aabb.buffer + .expect("needs buffer to build") + .resource + .GetGPUVirtualAddress() + + (aabb.offset as u64 * aabb.stride) + }; + + let aabb_desc = Direct3D12::D3D12_RAYTRACING_GEOMETRY_AABBS_DESC { + AABBCount: aabb.count as u64, + AABBs: Direct3D12::D3D12_GPU_VIRTUAL_ADDRESS_AND_STRIDE { + StartAddress: aabb_address, + StrideInBytes: aabb.stride, + }, + }; + + geometry_desc.push(Direct3D12::D3D12_RAYTRACING_GEOMETRY_DESC { + Type: Direct3D12::D3D12_RAYTRACING_GEOMETRY_TYPE_PROCEDURAL_PRIMITIVE_AABBS, + Flags: conv::map_acceleration_structure_geometry_flags(aabb.flags), + Anonymous: Direct3D12::D3D12_RAYTRACING_GEOMETRY_DESC_0 { + AABBs: aabb_desc, + }, + }) + } + ty = Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_TYPE_BOTTOM_LEVEL; + inputs0 = Direct3D12::D3D12_BUILD_RAYTRACING_ACCELERATION_STRUCTURE_INPUTS_0 { + pGeometryDescs: geometry_desc.as_ptr(), + }; + num_desc = geometry_desc.len() as u32; + } + }; + let acceleration_structure_inputs = + Direct3D12::D3D12_BUILD_RAYTRACING_ACCELERATION_STRUCTURE_INPUTS { + Type: ty, + Flags: conv::map_acceleration_structure_build_flags( + descriptor.flags, + Some(descriptor.mode), + ), + NumDescs: num_desc, + DescsLayout: Direct3D12::D3D12_ELEMENTS_LAYOUT_ARRAY, + Anonymous: inputs0, + }; + + let dst_acceleration_structure_address = unsafe { + descriptor + .destination_acceleration_structure + .resource + .GetGPUVirtualAddress() + }; + let src_acceleration_structure_address = descriptor + .source_acceleration_structure + .as_ref() + .map_or(0, |source| unsafe { + source.resource.GetGPUVirtualAddress() + }); + let scratch_address = unsafe { + descriptor.scratch_buffer.resource.GetGPUVirtualAddress() + + descriptor.scratch_buffer_offset + }; + + let desc = Direct3D12::D3D12_BUILD_RAYTRACING_ACCELERATION_STRUCTURE_DESC { + DestAccelerationStructureData: dst_acceleration_structure_address, + Inputs: acceleration_structure_inputs, + SourceAccelerationStructureData: src_acceleration_structure_address, + ScratchAccelerationStructureData: scratch_address, + }; + unsafe { list.BuildRaytracingAccelerationStructure(&desc, None) }; + } + } + + unsafe fn place_acceleration_structure_barrier( + &mut self, + _barriers: crate::AccelerationStructureBarrier, + ) { + // TODO: This is not very optimal, we should be using [enhanced barriers](https://microsoft.github.io/DirectX-Specs/d3d/D3D12EnhancedBarriers.html) if possible + let list = self + .list + .as_ref() + .unwrap() + .cast::() + .unwrap(); + unsafe { + list.ResourceBarrier(&[Direct3D12::D3D12_RESOURCE_BARRIER { + Type: Direct3D12::D3D12_RESOURCE_BARRIER_TYPE_UAV, + Flags: Direct3D12::D3D12_RESOURCE_BARRIER_FLAG_NONE, + Anonymous: Direct3D12::D3D12_RESOURCE_BARRIER_0 { + UAV: mem::ManuallyDrop::new(Direct3D12::D3D12_RESOURCE_UAV_BARRIER { + pResource: Default::default(), + }), + }, + }]) + } + } + + unsafe fn copy_acceleration_structure_to_acceleration_structure( + &mut self, + src: &super::AccelerationStructure, + dst: &super::AccelerationStructure, + copy: wgt::AccelerationStructureCopy, + ) { + let list = self + .list + .as_ref() + .unwrap() + .cast::() + .unwrap(); + unsafe { + list.CopyRaytracingAccelerationStructure( + dst.resource.GetGPUVirtualAddress(), + src.resource.GetGPUVirtualAddress(), + conv::map_acceleration_structure_copy_mode(copy), + ) + } + } + + unsafe fn set_acceleration_structure_dependencies( + _command_buffers: &[&super::CommandBuffer], + _dependencies: &[&super::AccelerationStructure], + ) { + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/conv.rs b/third_party/wgpu-hal-29.0.4/src/dx12/conv.rs new file mode 100644 index 0000000..064700b --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/conv.rs @@ -0,0 +1,449 @@ +use windows::Win32::Graphics::{Direct3D, Direct3D12, Dxgi}; + +pub fn map_buffer_usage_to_resource_flags( + usage: wgt::BufferUses, +) -> Direct3D12::D3D12_RESOURCE_FLAGS { + let mut flags = Direct3D12::D3D12_RESOURCE_FLAG_NONE; + if usage.contains(wgt::BufferUses::STORAGE_READ_WRITE) + || usage.contains(wgt::BufferUses::ACCELERATION_STRUCTURE_QUERY) + { + flags |= Direct3D12::D3D12_RESOURCE_FLAG_ALLOW_UNORDERED_ACCESS; + } + flags +} + +pub fn map_buffer_descriptor( + desc: &crate::BufferDescriptor<'_>, +) -> Direct3D12::D3D12_RESOURCE_DESC { + Direct3D12::D3D12_RESOURCE_DESC { + Dimension: Direct3D12::D3D12_RESOURCE_DIMENSION_BUFFER, + Alignment: 0, + Width: desc.size, + Height: 1, + DepthOrArraySize: 1, + MipLevels: 1, + Format: Dxgi::Common::DXGI_FORMAT_UNKNOWN, + SampleDesc: Dxgi::Common::DXGI_SAMPLE_DESC { + Count: 1, + Quality: 0, + }, + Layout: Direct3D12::D3D12_TEXTURE_LAYOUT_ROW_MAJOR, + Flags: map_buffer_usage_to_resource_flags(desc.usage), + } +} + +pub fn map_texture_dimension(dim: wgt::TextureDimension) -> Direct3D12::D3D12_RESOURCE_DIMENSION { + match dim { + wgt::TextureDimension::D1 => Direct3D12::D3D12_RESOURCE_DIMENSION_TEXTURE1D, + wgt::TextureDimension::D2 => Direct3D12::D3D12_RESOURCE_DIMENSION_TEXTURE2D, + wgt::TextureDimension::D3 => Direct3D12::D3D12_RESOURCE_DIMENSION_TEXTURE3D, + } +} + +pub fn map_texture_usage_to_resource_flags( + usage: wgt::TextureUses, +) -> Direct3D12::D3D12_RESOURCE_FLAGS { + let mut flags = Direct3D12::D3D12_RESOURCE_FLAG_NONE; + + if usage.contains(wgt::TextureUses::COLOR_TARGET) { + flags |= Direct3D12::D3D12_RESOURCE_FLAG_ALLOW_RENDER_TARGET; + } + if usage + .intersects(wgt::TextureUses::DEPTH_STENCIL_READ | wgt::TextureUses::DEPTH_STENCIL_WRITE) + { + flags |= Direct3D12::D3D12_RESOURCE_FLAG_ALLOW_DEPTH_STENCIL; + if !usage.contains(wgt::TextureUses::RESOURCE) { + flags |= Direct3D12::D3D12_RESOURCE_FLAG_DENY_SHADER_RESOURCE; + } + } + if usage.intersects( + wgt::TextureUses::STORAGE_READ_ONLY + | wgt::TextureUses::STORAGE_WRITE_ONLY + | wgt::TextureUses::STORAGE_READ_WRITE, + ) { + flags |= Direct3D12::D3D12_RESOURCE_FLAG_ALLOW_UNORDERED_ACCESS; + } + + flags +} + +pub fn map_address_mode(mode: wgt::AddressMode) -> Direct3D12::D3D12_TEXTURE_ADDRESS_MODE { + use wgt::AddressMode as Am; + match mode { + Am::Repeat => Direct3D12::D3D12_TEXTURE_ADDRESS_MODE_WRAP, + Am::MirrorRepeat => Direct3D12::D3D12_TEXTURE_ADDRESS_MODE_MIRROR, + Am::ClampToEdge => Direct3D12::D3D12_TEXTURE_ADDRESS_MODE_CLAMP, + Am::ClampToBorder => Direct3D12::D3D12_TEXTURE_ADDRESS_MODE_BORDER, + //Am::MirrorClamp => Direct3D12::D3D12_TEXTURE_ADDRESS_MODE_MIRROR_ONCE, + } +} + +pub fn map_filter_mode(mode: wgt::FilterMode) -> Direct3D12::D3D12_FILTER_TYPE { + match mode { + wgt::FilterMode::Nearest => Direct3D12::D3D12_FILTER_TYPE_POINT, + wgt::FilterMode::Linear => Direct3D12::D3D12_FILTER_TYPE_LINEAR, + } +} + +pub fn map_mipmap_filter_mode(mode: wgt::MipmapFilterMode) -> Direct3D12::D3D12_FILTER_TYPE { + match mode { + wgt::MipmapFilterMode::Nearest => Direct3D12::D3D12_FILTER_TYPE_POINT, + wgt::MipmapFilterMode::Linear => Direct3D12::D3D12_FILTER_TYPE_LINEAR, + } +} + +pub fn map_comparison(func: wgt::CompareFunction) -> Direct3D12::D3D12_COMPARISON_FUNC { + use wgt::CompareFunction as Cf; + match func { + Cf::Never => Direct3D12::D3D12_COMPARISON_FUNC_NEVER, + Cf::Less => Direct3D12::D3D12_COMPARISON_FUNC_LESS, + Cf::LessEqual => Direct3D12::D3D12_COMPARISON_FUNC_LESS_EQUAL, + Cf::Equal => Direct3D12::D3D12_COMPARISON_FUNC_EQUAL, + Cf::GreaterEqual => Direct3D12::D3D12_COMPARISON_FUNC_GREATER_EQUAL, + Cf::Greater => Direct3D12::D3D12_COMPARISON_FUNC_GREATER, + Cf::NotEqual => Direct3D12::D3D12_COMPARISON_FUNC_NOT_EQUAL, + Cf::Always => Direct3D12::D3D12_COMPARISON_FUNC_ALWAYS, + } +} + +pub fn map_border_color(border_color: Option) -> [f32; 4] { + use wgt::SamplerBorderColor as Sbc; + match border_color { + Some(Sbc::TransparentBlack) | Some(Sbc::Zero) | None => [0.0; 4], + Some(Sbc::OpaqueBlack) => [0.0, 0.0, 0.0, 1.0], + Some(Sbc::OpaqueWhite) => [1.0; 4], + } +} + +pub fn map_visibility(visibility: wgt::ShaderStages) -> Direct3D12::D3D12_SHADER_VISIBILITY { + match visibility { + wgt::ShaderStages::VERTEX => Direct3D12::D3D12_SHADER_VISIBILITY_VERTEX, + wgt::ShaderStages::FRAGMENT => Direct3D12::D3D12_SHADER_VISIBILITY_PIXEL, + _ => Direct3D12::D3D12_SHADER_VISIBILITY_ALL, + } +} + +pub fn map_binding_type(ty: &wgt::BindingType) -> Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE { + use wgt::BindingType as Bt; + match *ty { + Bt::Sampler { .. } => Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE_SAMPLER, + Bt::Buffer { + ty: wgt::BufferBindingType::Uniform, + .. + } => Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE_CBV, + Bt::Buffer { + ty: wgt::BufferBindingType::Storage { read_only: true }, + .. + } + | Bt::Texture { .. } => Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE_SRV, + Bt::Buffer { + ty: wgt::BufferBindingType::Storage { read_only: false }, + .. + } + | Bt::StorageTexture { .. } => Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE_UAV, + Bt::AccelerationStructure { .. } => Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE_SRV, + // External textures require multiple bindings and therefore cannot + // be mapped to a single descriptor range type. They must be handled + // separately by the caller. + Bt::ExternalTexture => unreachable!("External textures must be handled separately"), + } +} + +pub fn map_buffer_usage_to_state(usage: wgt::BufferUses) -> Direct3D12::D3D12_RESOURCE_STATES { + use wgt::BufferUses as Bu; + let mut state = Direct3D12::D3D12_RESOURCE_STATE_COMMON; + + if usage.intersects(Bu::COPY_SRC) { + state |= Direct3D12::D3D12_RESOURCE_STATE_COPY_SOURCE; + } + if usage.intersects(Bu::COPY_DST) { + state |= Direct3D12::D3D12_RESOURCE_STATE_COPY_DEST; + } + if usage.intersects(Bu::INDEX) { + state |= Direct3D12::D3D12_RESOURCE_STATE_INDEX_BUFFER; + } + if usage.intersects(Bu::VERTEX | Bu::UNIFORM) { + state |= Direct3D12::D3D12_RESOURCE_STATE_VERTEX_AND_CONSTANT_BUFFER; + } + if usage.intersects(Bu::STORAGE_READ_WRITE) { + state |= Direct3D12::D3D12_RESOURCE_STATE_UNORDERED_ACCESS; + } else if usage.intersects(Bu::STORAGE_READ_ONLY) { + state |= Direct3D12::D3D12_RESOURCE_STATE_PIXEL_SHADER_RESOURCE + | Direct3D12::D3D12_RESOURCE_STATE_NON_PIXEL_SHADER_RESOURCE; + } + if usage.intersects(Bu::INDIRECT) { + state |= Direct3D12::D3D12_RESOURCE_STATE_INDIRECT_ARGUMENT; + } + if usage.intersects(Bu::ACCELERATION_STRUCTURE_QUERY) { + state |= Direct3D12::D3D12_RESOURCE_STATE_UNORDERED_ACCESS; + } + state +} + +pub fn map_texture_usage_to_state(usage: wgt::TextureUses) -> Direct3D12::D3D12_RESOURCE_STATES { + use wgt::TextureUses as Tu; + let mut state = Direct3D12::D3D12_RESOURCE_STATE_COMMON; + //Note: `RESOLVE_SOURCE` and `RESOLVE_DEST` are not used here + //Note: `PRESENT` is the same as `COMMON` + if usage == wgt::TextureUses::UNINITIALIZED { + return state; + } + + if usage.intersects(Tu::COPY_SRC) { + state |= Direct3D12::D3D12_RESOURCE_STATE_COPY_SOURCE; + } + if usage.intersects(Tu::COPY_DST) { + state |= Direct3D12::D3D12_RESOURCE_STATE_COPY_DEST; + } + if usage.intersects(Tu::RESOURCE) { + state |= Direct3D12::D3D12_RESOURCE_STATE_PIXEL_SHADER_RESOURCE + | Direct3D12::D3D12_RESOURCE_STATE_NON_PIXEL_SHADER_RESOURCE; + } + if usage.intersects(Tu::COLOR_TARGET) { + state |= Direct3D12::D3D12_RESOURCE_STATE_RENDER_TARGET; + } + if usage.intersects(Tu::DEPTH_STENCIL_READ) { + state |= Direct3D12::D3D12_RESOURCE_STATE_DEPTH_READ; + } + if usage.intersects(Tu::DEPTH_STENCIL_WRITE) { + state |= Direct3D12::D3D12_RESOURCE_STATE_DEPTH_WRITE; + } + if usage.intersects(Tu::STORAGE_READ_ONLY | Tu::STORAGE_WRITE_ONLY | Tu::STORAGE_READ_WRITE) { + state |= Direct3D12::D3D12_RESOURCE_STATE_UNORDERED_ACCESS; + } + state +} + +pub fn map_topology( + topology: wgt::PrimitiveTopology, +) -> ( + Direct3D12::D3D12_PRIMITIVE_TOPOLOGY_TYPE, + Direct3D::D3D_PRIMITIVE_TOPOLOGY, +) { + match topology { + wgt::PrimitiveTopology::PointList => ( + Direct3D12::D3D12_PRIMITIVE_TOPOLOGY_TYPE_POINT, + Direct3D::D3D_PRIMITIVE_TOPOLOGY_POINTLIST, + ), + wgt::PrimitiveTopology::LineList => ( + Direct3D12::D3D12_PRIMITIVE_TOPOLOGY_TYPE_LINE, + Direct3D::D3D_PRIMITIVE_TOPOLOGY_LINELIST, + ), + wgt::PrimitiveTopology::LineStrip => ( + Direct3D12::D3D12_PRIMITIVE_TOPOLOGY_TYPE_LINE, + Direct3D::D3D_PRIMITIVE_TOPOLOGY_LINESTRIP, + ), + wgt::PrimitiveTopology::TriangleList => ( + Direct3D12::D3D12_PRIMITIVE_TOPOLOGY_TYPE_TRIANGLE, + Direct3D::D3D_PRIMITIVE_TOPOLOGY_TRIANGLELIST, + ), + wgt::PrimitiveTopology::TriangleStrip => ( + Direct3D12::D3D12_PRIMITIVE_TOPOLOGY_TYPE_TRIANGLE, + Direct3D::D3D_PRIMITIVE_TOPOLOGY_TRIANGLESTRIP, + ), + } +} + +pub fn map_polygon_mode(mode: wgt::PolygonMode) -> Direct3D12::D3D12_FILL_MODE { + match mode { + wgt::PolygonMode::Fill => Direct3D12::D3D12_FILL_MODE_SOLID, + wgt::PolygonMode::Line => Direct3D12::D3D12_FILL_MODE_WIREFRAME, + wgt::PolygonMode::Point => panic!( + "{:?} is not enabled for this backend", + wgt::Features::POLYGON_MODE_POINT + ), + } +} + +/// D3D12 doesn't support passing factors ending in `_COLOR` for alpha blending +/// (see ). +/// Therefore this function takes an additional `is_alpha` argument +/// which if set will return an equivalent `_ALPHA` factor. +fn map_blend_factor(factor: wgt::BlendFactor, is_alpha: bool) -> Direct3D12::D3D12_BLEND { + use wgt::BlendFactor as Bf; + match factor { + Bf::Zero => Direct3D12::D3D12_BLEND_ZERO, + Bf::One => Direct3D12::D3D12_BLEND_ONE, + Bf::Src if is_alpha => Direct3D12::D3D12_BLEND_SRC_ALPHA, + Bf::Src => Direct3D12::D3D12_BLEND_SRC_COLOR, + Bf::OneMinusSrc if is_alpha => Direct3D12::D3D12_BLEND_INV_SRC_ALPHA, + Bf::OneMinusSrc => Direct3D12::D3D12_BLEND_INV_SRC_COLOR, + Bf::Dst if is_alpha => Direct3D12::D3D12_BLEND_DEST_ALPHA, + Bf::Dst => Direct3D12::D3D12_BLEND_DEST_COLOR, + Bf::OneMinusDst if is_alpha => Direct3D12::D3D12_BLEND_INV_DEST_ALPHA, + Bf::OneMinusDst => Direct3D12::D3D12_BLEND_INV_DEST_COLOR, + Bf::SrcAlpha => Direct3D12::D3D12_BLEND_SRC_ALPHA, + Bf::OneMinusSrcAlpha => Direct3D12::D3D12_BLEND_INV_SRC_ALPHA, + Bf::DstAlpha => Direct3D12::D3D12_BLEND_DEST_ALPHA, + Bf::OneMinusDstAlpha => Direct3D12::D3D12_BLEND_INV_DEST_ALPHA, + Bf::Constant => Direct3D12::D3D12_BLEND_BLEND_FACTOR, + Bf::OneMinusConstant => Direct3D12::D3D12_BLEND_INV_BLEND_FACTOR, + Bf::SrcAlphaSaturated => Direct3D12::D3D12_BLEND_SRC_ALPHA_SAT, + Bf::Src1 if is_alpha => Direct3D12::D3D12_BLEND_SRC1_ALPHA, + Bf::Src1 => Direct3D12::D3D12_BLEND_SRC1_COLOR, + Bf::OneMinusSrc1 if is_alpha => Direct3D12::D3D12_BLEND_INV_SRC1_ALPHA, + Bf::OneMinusSrc1 => Direct3D12::D3D12_BLEND_INV_SRC1_COLOR, + Bf::Src1Alpha => Direct3D12::D3D12_BLEND_SRC1_ALPHA, + Bf::OneMinusSrc1Alpha => Direct3D12::D3D12_BLEND_INV_SRC1_ALPHA, + } +} + +fn map_blend_component( + component: &wgt::BlendComponent, + is_alpha: bool, +) -> ( + Direct3D12::D3D12_BLEND_OP, + Direct3D12::D3D12_BLEND, + Direct3D12::D3D12_BLEND, +) { + let raw_op = match component.operation { + wgt::BlendOperation::Add => Direct3D12::D3D12_BLEND_OP_ADD, + wgt::BlendOperation::Subtract => Direct3D12::D3D12_BLEND_OP_SUBTRACT, + wgt::BlendOperation::ReverseSubtract => Direct3D12::D3D12_BLEND_OP_REV_SUBTRACT, + wgt::BlendOperation::Min => Direct3D12::D3D12_BLEND_OP_MIN, + wgt::BlendOperation::Max => Direct3D12::D3D12_BLEND_OP_MAX, + }; + let raw_src = map_blend_factor(component.src_factor, is_alpha); + let raw_dst = map_blend_factor(component.dst_factor, is_alpha); + (raw_op, raw_src, raw_dst) +} + +pub fn map_render_targets( + color_targets: &[Option], +) -> [Direct3D12::D3D12_RENDER_TARGET_BLEND_DESC; + Direct3D12::D3D12_SIMULTANEOUS_RENDER_TARGET_COUNT as usize] { + let dummy_target = Direct3D12::D3D12_RENDER_TARGET_BLEND_DESC { + BlendEnable: false.into(), + LogicOpEnable: false.into(), + SrcBlend: Direct3D12::D3D12_BLEND_ZERO, + DestBlend: Direct3D12::D3D12_BLEND_ZERO, + BlendOp: Direct3D12::D3D12_BLEND_OP_ADD, + SrcBlendAlpha: Direct3D12::D3D12_BLEND_ZERO, + DestBlendAlpha: Direct3D12::D3D12_BLEND_ZERO, + BlendOpAlpha: Direct3D12::D3D12_BLEND_OP_ADD, + LogicOp: Direct3D12::D3D12_LOGIC_OP_CLEAR, + RenderTargetWriteMask: 0, + }; + let mut raw_targets = + [dummy_target; Direct3D12::D3D12_SIMULTANEOUS_RENDER_TARGET_COUNT as usize]; + + for (raw, ct) in raw_targets.iter_mut().zip(color_targets.iter()) { + if let Some(ct) = ct.as_ref() { + raw.RenderTargetWriteMask = ct.write_mask.bits() as u8; + if let Some(ref blend) = ct.blend { + let (color_op, color_src, color_dst) = map_blend_component(&blend.color, false); + let (alpha_op, alpha_src, alpha_dst) = map_blend_component(&blend.alpha, true); + raw.BlendEnable = true.into(); + raw.BlendOp = color_op; + raw.SrcBlend = color_src; + raw.DestBlend = color_dst; + raw.BlendOpAlpha = alpha_op; + raw.SrcBlendAlpha = alpha_src; + raw.DestBlendAlpha = alpha_dst; + } + } + } + + raw_targets +} + +fn map_stencil_op(op: wgt::StencilOperation) -> Direct3D12::D3D12_STENCIL_OP { + use wgt::StencilOperation as So; + match op { + So::Keep => Direct3D12::D3D12_STENCIL_OP_KEEP, + So::Zero => Direct3D12::D3D12_STENCIL_OP_ZERO, + So::Replace => Direct3D12::D3D12_STENCIL_OP_REPLACE, + So::IncrementClamp => Direct3D12::D3D12_STENCIL_OP_INCR_SAT, + So::IncrementWrap => Direct3D12::D3D12_STENCIL_OP_INCR, + So::DecrementClamp => Direct3D12::D3D12_STENCIL_OP_DECR_SAT, + So::DecrementWrap => Direct3D12::D3D12_STENCIL_OP_DECR, + So::Invert => Direct3D12::D3D12_STENCIL_OP_INVERT, + } +} + +fn map_stencil_face(face: &wgt::StencilFaceState) -> Direct3D12::D3D12_DEPTH_STENCILOP_DESC { + Direct3D12::D3D12_DEPTH_STENCILOP_DESC { + StencilFailOp: map_stencil_op(face.fail_op), + StencilDepthFailOp: map_stencil_op(face.depth_fail_op), + StencilPassOp: map_stencil_op(face.pass_op), + StencilFunc: map_comparison(face.compare), + } +} + +pub fn map_depth_stencil(ds: &wgt::DepthStencilState) -> Direct3D12::D3D12_DEPTH_STENCIL_DESC { + Direct3D12::D3D12_DEPTH_STENCIL_DESC { + DepthEnable: ds.is_depth_enabled().into(), + DepthWriteMask: if ds.depth_write_enabled.unwrap_or_default() { + Direct3D12::D3D12_DEPTH_WRITE_MASK_ALL + } else { + Direct3D12::D3D12_DEPTH_WRITE_MASK_ZERO + }, + DepthFunc: map_comparison(ds.depth_compare.unwrap_or_default()), + StencilEnable: ds.stencil.is_enabled().into(), + StencilReadMask: ds.stencil.read_mask as u8, + StencilWriteMask: ds.stencil.write_mask as u8, + FrontFace: map_stencil_face(&ds.stencil.front), + BackFace: map_stencil_face(&ds.stencil.back), + } +} + +pub(crate) fn map_acceleration_structure_build_flags( + flags: wgt::AccelerationStructureFlags, + mode: Option, +) -> Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_BUILD_FLAGS { + let mut d3d_flags = Default::default(); + if flags.contains(wgt::AccelerationStructureFlags::ALLOW_COMPACTION) { + d3d_flags |= + Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_BUILD_FLAG_ALLOW_COMPACTION; + } + + if flags.contains(wgt::AccelerationStructureFlags::ALLOW_UPDATE) { + d3d_flags |= Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_BUILD_FLAG_ALLOW_UPDATE; + } + + if flags.contains(wgt::AccelerationStructureFlags::LOW_MEMORY) { + d3d_flags |= Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_BUILD_FLAG_MINIMIZE_MEMORY; + } + + if flags.contains(wgt::AccelerationStructureFlags::PREFER_FAST_BUILD) { + d3d_flags |= + Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_BUILD_FLAG_PREFER_FAST_BUILD; + } + + if flags.contains(wgt::AccelerationStructureFlags::PREFER_FAST_TRACE) { + d3d_flags |= + Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_BUILD_FLAG_PREFER_FAST_TRACE; + } + + if let Some(crate::AccelerationStructureBuildMode::Update) = mode { + d3d_flags |= Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_BUILD_FLAG_PERFORM_UPDATE + } + + d3d_flags +} + +pub(crate) fn map_acceleration_structure_geometry_flags( + flags: wgt::AccelerationStructureGeometryFlags, +) -> Direct3D12::D3D12_RAYTRACING_GEOMETRY_FLAGS { + let mut d3d_flags = Default::default(); + if flags.contains(wgt::AccelerationStructureGeometryFlags::OPAQUE) { + d3d_flags |= Direct3D12::D3D12_RAYTRACING_GEOMETRY_FLAG_OPAQUE; + } + if flags.contains(wgt::AccelerationStructureGeometryFlags::NO_DUPLICATE_ANY_HIT_INVOCATION) { + d3d_flags |= Direct3D12::D3D12_RAYTRACING_GEOMETRY_FLAG_NO_DUPLICATE_ANYHIT_INVOCATION; + } + d3d_flags +} + +pub(crate) fn map_acceleration_structure_copy_mode( + mode: wgt::AccelerationStructureCopy, +) -> Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_COPY_MODE { + match mode { + wgt::AccelerationStructureCopy::Clone => { + Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_COPY_MODE_CLONE + } + wgt::AccelerationStructureCopy::Compact => { + Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_COPY_MODE_COMPACT + } + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/dcomp.rs b/third_party/wgpu-hal-29.0.4/src/dx12/dcomp.rs new file mode 100644 index 0000000..46ef3da --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/dcomp.rs @@ -0,0 +1,120 @@ +use alloc::sync::Arc; +use core::{ffi, ptr}; + +use once_cell::sync::Lazy; +use windows::{ + core::Interface as _, + Win32::{Foundation::HWND, Graphics::DirectComposition}, +}; + +use super::DynLib; + +// Lazy-loaded DirectComposition library +#[derive(Debug)] +pub(crate) struct DCompLib { + lib: Lazy>, +} + +impl DCompLib { + pub(crate) fn new() -> Self { + Self { + lib: Lazy::new(|| unsafe { + DynLib::new("dcomp.dll").map_err(|err| { + log::error!("Error loading dcomp.dll: {err}"); + crate::SurfaceError::Other("Error loading dcomp.dll") + }) + }), + } + } + + fn get_lib(&self) -> Result<&DynLib, crate::SurfaceError> { + match self.lib.as_ref() { + Ok(lib) => Ok(lib), + Err(err) => Err(err.clone()), + } + } + + pub(crate) fn create_device( + &self, + ) -> Result { + let lib = self.get_lib()?; + + // Calls windows::Win32::Graphics::DirectComposition::DCompositionCreateDevice2 on dcomp.dll + type Fun = extern "system" fn( + pdxdevice: *mut ffi::c_void, + riid: *const windows_core::GUID, + ppdcompdevice: *mut *mut ffi::c_void, + ) -> windows_core::HRESULT; + let func: libloading::Symbol = + unsafe { lib.get(c"DCompositionCreateDevice2".to_bytes()) }?; + + let mut res: Option = None; + + (func)( + ptr::null_mut(), + &DirectComposition::IDCompositionDevice::IID, + <*mut _>::cast(&mut res), + ) + .map(|| res.unwrap()) + .map_err(|err| { + log::error!("DirectComposition::DCompositionCreateDevice2 failed: {err}"); + crate::SurfaceError::Other("DirectComposition::DCompositionCreateDevice2") + }) + } +} + +#[derive(Default)] +pub struct DCompState { + inner: Option, +} + +impl DCompState { + /// This will create a DirectComposition device and a target for the window handle if not already initialized. + /// If the device is already initialized, it will return the existing state. + pub unsafe fn get_or_init( + &mut self, + lib: &Arc, + hwnd: &HWND, + ) -> Result<&mut InnerState, crate::SurfaceError> { + if self.inner.is_none() { + self.inner = Some(unsafe { InnerState::init(lib, hwnd) }?); + } + Ok(self.inner.as_mut().unwrap()) + } +} + +pub struct InnerState { + pub visual: DirectComposition::IDCompositionVisual, + pub device: DirectComposition::IDCompositionDevice, + // Must be kept alive but is otherwise unused after initialization. + pub _target: DirectComposition::IDCompositionTarget, +} + +impl InnerState { + /// Creates a DirectComposition device and a target for the given window handle. + pub unsafe fn init(lib: &Arc, hwnd: &HWND) -> Result { + profiling::scope!("DCompState::init"); + let dcomp_device = lib.create_device()?; + + let target = unsafe { dcomp_device.CreateTargetForHwnd(*hwnd, false) }.map_err(|err| { + log::error!("IDCompositionDevice::CreateTargetForHwnd failed: {err}"); + crate::SurfaceError::Other("IDCompositionDevice::CreateTargetForHwnd") + })?; + + let visual = unsafe { dcomp_device.CreateVisual() }.map_err(|err| { + log::error!("IDCompositionDevice::CreateVisual failed: {err}"); + crate::SurfaceError::Other("IDCompositionDevice::CreateVisual") + })?; + + unsafe { target.SetRoot(&visual) }.map_err(|err| { + log::error!("IDCompositionTarget::SetRoot failed: {err}"); + crate::SurfaceError::Other("IDCompositionTarget::SetRoot") + })?; + + Ok(InnerState { + visual, + device: dcomp_device, + _target: target, + }) + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/descriptor.rs b/third_party/wgpu-hal-29.0.4/src/dx12/descriptor.rs new file mode 100644 index 0000000..f57ba46 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/descriptor.rs @@ -0,0 +1,329 @@ +use alloc::vec::Vec; +use core::fmt; + +use bit_set::BitSet; +use parking_lot::Mutex; +use range_alloc::RangeAllocator; +use windows::Win32::Graphics::Direct3D12; + +use crate::auxil::dxgi::result::HResult as _; + +const HEAP_SIZE_FIXED: usize = 64; + +#[derive(Copy, Clone)] +pub(super) struct DualHandle { + cpu: Direct3D12::D3D12_CPU_DESCRIPTOR_HANDLE, + pub gpu: Direct3D12::D3D12_GPU_DESCRIPTOR_HANDLE, + /// How large the block allocated to this handle is. + count: u64, +} + +impl fmt::Debug for DualHandle { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("DualHandle") + .field("cpu", &self.cpu.ptr) + .field("gpu", &self.gpu.ptr) + .field("count", &self.count) + .finish() + } +} + +type DescriptorIndex = u64; + +pub(super) struct GeneralHeap { + pub raw: Direct3D12::ID3D12DescriptorHeap, + ty: Direct3D12::D3D12_DESCRIPTOR_HEAP_TYPE, + handle_size: u64, + total_handles: u64, + start: DualHandle, + ranges: Mutex>, +} + +impl GeneralHeap { + pub(super) fn new( + device: &Direct3D12::ID3D12Device, + ty: Direct3D12::D3D12_DESCRIPTOR_HEAP_TYPE, + total_handles: u64, + ) -> Result { + let raw = { + profiling::scope!("ID3D12Device::CreateDescriptorHeap"); + let desc = Direct3D12::D3D12_DESCRIPTOR_HEAP_DESC { + Type: ty, + NumDescriptors: total_handles as u32, + Flags: Direct3D12::D3D12_DESCRIPTOR_HEAP_FLAG_SHADER_VISIBLE, + NodeMask: 0, + }; + unsafe { device.CreateDescriptorHeap::(&desc) } + .into_device_result("Descriptor heap creation")? + }; + + let start = DualHandle { + cpu: unsafe { raw.GetCPUDescriptorHandleForHeapStart() }, + gpu: unsafe { raw.GetGPUDescriptorHandleForHeapStart() }, + count: 0, + }; + + Ok(Self { + raw, + ty, + handle_size: unsafe { device.GetDescriptorHandleIncrementSize(ty) } as u64, + total_handles, + start, + ranges: Mutex::new(RangeAllocator::new(0..total_handles)), + }) + } + + pub(super) fn at(&self, index: DescriptorIndex, count: u64) -> DualHandle { + assert!(index < self.total_handles); + DualHandle { + cpu: self.cpu_descriptor_at(index), + gpu: self.gpu_descriptor_at(index), + count, + } + } + + fn cpu_descriptor_at(&self, index: u64) -> Direct3D12::D3D12_CPU_DESCRIPTOR_HANDLE { + Direct3D12::D3D12_CPU_DESCRIPTOR_HANDLE { + ptr: self.start.cpu.ptr + (self.handle_size * index) as usize, + } + } + + fn gpu_descriptor_at(&self, index: u64) -> Direct3D12::D3D12_GPU_DESCRIPTOR_HANDLE { + Direct3D12::D3D12_GPU_DESCRIPTOR_HANDLE { + ptr: self.start.gpu.ptr + self.handle_size * index, + } + } + + pub(super) fn allocate_slice(&self, count: u64) -> Result { + let range = self.ranges.lock().allocate_range(count).map_err(|err| { + log::error!("Unable to allocate descriptors: {err:?}"); + crate::DeviceError::OutOfMemory + })?; + Ok(range.start) + } + + /// Free handles previously given out by this `DescriptorHeapSlice`. + /// Do not use this with handles not given out by this `DescriptorHeapSlice`. + pub(crate) fn free_slice(&self, handle: DualHandle) { + let start = (handle.gpu.ptr - self.start.gpu.ptr) / self.handle_size; + self.ranges.lock().free_range(start..start + handle.count); + } +} + +/// Fixed-size free-list allocator for CPU descriptors. +struct FixedSizeHeap { + _raw: Direct3D12::ID3D12DescriptorHeap, + /// Bit flag representation of available handles in the heap. + /// + /// 0 - Occupied + /// 1 - free + availability: u64, + handle_size: usize, + start: Direct3D12::D3D12_CPU_DESCRIPTOR_HANDLE, +} + +impl FixedSizeHeap { + fn new( + device: &Direct3D12::ID3D12Device, + ty: Direct3D12::D3D12_DESCRIPTOR_HEAP_TYPE, + ) -> Result { + let desc = Direct3D12::D3D12_DESCRIPTOR_HEAP_DESC { + Type: ty, + NumDescriptors: HEAP_SIZE_FIXED as u32, + Flags: Direct3D12::D3D12_DESCRIPTOR_HEAP_FLAG_NONE, + NodeMask: 0, + }; + let heap = + unsafe { device.CreateDescriptorHeap::(&desc) } + .into_device_result("Descriptor heap creation")?; + + Ok(Self { + handle_size: unsafe { device.GetDescriptorHandleIncrementSize(ty) } as usize, + availability: !0, // all free! + start: unsafe { heap.GetCPUDescriptorHandleForHeapStart() }, + _raw: heap, + }) + } + + fn alloc_handle( + &mut self, + ) -> Result { + // Find first free slot. + let slot = self.availability.trailing_zeros() as usize; + if slot >= HEAP_SIZE_FIXED { + log::error!("Failed to allocate a handle form a fixed size heap"); + return Err(crate::DeviceError::OutOfMemory); + } + // Set the slot as occupied. + self.availability ^= 1 << slot; + + Ok(Direct3D12::D3D12_CPU_DESCRIPTOR_HANDLE { + ptr: self.start.ptr + self.handle_size * slot, + }) + } + + fn free_handle(&mut self, handle: Direct3D12::D3D12_CPU_DESCRIPTOR_HANDLE) { + let slot = (handle.ptr - self.start.ptr) / self.handle_size; + assert!(slot < HEAP_SIZE_FIXED); + assert_eq!(self.availability & (1 << slot), 0); + self.availability ^= 1 << slot; + } + + fn is_full(&self) -> bool { + self.availability == 0 + } +} + +#[derive(Clone, Copy)] +pub(super) struct Handle { + pub raw: Direct3D12::D3D12_CPU_DESCRIPTOR_HANDLE, + heap_index: usize, +} + +impl fmt::Debug for Handle { + fn fmt(&self, fmt: &mut fmt::Formatter) -> fmt::Result { + fmt.debug_struct("Handle") + .field("ptr", &self.raw.ptr) + .field("heap_index", &self.heap_index) + .finish() + } +} + +pub(super) struct CpuPool { + device: Direct3D12::ID3D12Device, + ty: Direct3D12::D3D12_DESCRIPTOR_HEAP_TYPE, + heaps: Vec, + available_heap_indices: BitSet, +} + +impl CpuPool { + pub(super) fn new( + device: Direct3D12::ID3D12Device, + ty: Direct3D12::D3D12_DESCRIPTOR_HEAP_TYPE, + ) -> Self { + Self { + device, + ty, + heaps: Vec::new(), + available_heap_indices: BitSet::new(), + } + } + + pub(super) fn alloc_handle(&mut self) -> Result { + let heap_index = self + .available_heap_indices + .iter() + .next() + .unwrap_or(self.heaps.len()); + + // Allocate a new heap + if heap_index == self.heaps.len() { + self.heaps.push(FixedSizeHeap::new(&self.device, self.ty)?); + self.available_heap_indices.insert(heap_index); + } + + let heap = &mut self.heaps[heap_index]; + let handle = Handle { + raw: heap.alloc_handle()?, + heap_index, + }; + if heap.is_full() { + self.available_heap_indices.remove(heap_index); + } + + Ok(handle) + } + + pub(super) fn free_handle(&mut self, handle: Handle) { + self.heaps[handle.heap_index].free_handle(handle.raw); + self.available_heap_indices.insert(handle.heap_index); + } +} + +pub(super) struct CpuHeapInner { + pub _raw: Direct3D12::ID3D12DescriptorHeap, + pub stage: Vec, +} + +pub(super) struct CpuHeap { + pub inner: Mutex, + start: Direct3D12::D3D12_CPU_DESCRIPTOR_HANDLE, + handle_size: u32, + total: u32, +} + +unsafe impl Send for CpuHeap {} +unsafe impl Sync for CpuHeap {} + +impl CpuHeap { + pub(super) fn new( + device: &Direct3D12::ID3D12Device, + ty: Direct3D12::D3D12_DESCRIPTOR_HEAP_TYPE, + total: u32, + ) -> Result { + let handle_size = unsafe { device.GetDescriptorHandleIncrementSize(ty) }; + let desc = Direct3D12::D3D12_DESCRIPTOR_HEAP_DESC { + Type: ty, + NumDescriptors: total, + Flags: Direct3D12::D3D12_DESCRIPTOR_HEAP_FLAG_NONE, + NodeMask: 0, + }; + let raw = unsafe { device.CreateDescriptorHeap::(&desc) } + .into_device_result("CPU descriptor heap creation")?; + + let start = unsafe { raw.GetCPUDescriptorHandleForHeapStart() }; + + Ok(Self { + inner: Mutex::new(CpuHeapInner { + _raw: raw, + stage: Vec::new(), + }), + start, + handle_size, + total, + }) + } + + pub(super) fn at(&self, index: u32) -> Direct3D12::D3D12_CPU_DESCRIPTOR_HANDLE { + debug_assert!( + index < self.total, + "Index ({index}) out of bounds {total}", + total = self.total + ); + Direct3D12::D3D12_CPU_DESCRIPTOR_HANDLE { + ptr: self.start.ptr + (self.handle_size * index) as usize, + } + } +} + +impl fmt::Debug for CpuHeap { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("CpuHeap") + .field("start", &self.start.ptr) + .field("handle_size", &self.handle_size) + .field("total", &self.total) + .finish() + } +} + +pub(super) unsafe fn upload( + device: &Direct3D12::ID3D12Device, + src: &CpuHeapInner, + dst: &GeneralHeap, + dummy_copy_counts: &[u32], +) -> Result { + let count = src.stage.len() as u32; + let index = dst.allocate_slice(count as u64)?; + unsafe { + device.CopyDescriptors( + 1, + &dst.cpu_descriptor_at(index), + Some(&count), + count, + src.stage.as_ptr(), + Some(dummy_copy_counts.as_ptr()), + dst.ty, + ) + }; + Ok(dst.at(index, count as u64)) +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/device.rs b/third_party/wgpu-hal-29.0.4/src/dx12/device.rs new file mode 100644 index 0000000..5571d01 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/device.rs @@ -0,0 +1,2636 @@ +use alloc::borrow::ToOwned; +use alloc::{ + borrow::Cow, + string::{String, ToString as _}, + sync::Arc, + vec::Vec, +}; +use arrayvec::ArrayVec; +use core::{ffi, num::NonZeroU32, ptr, time::Duration}; +use std::time::Instant; + +use bytemuck::TransparentWrapper; +use parking_lot::Mutex; +use windows::{ + core::Interface as _, + Win32::{ + Foundation, + Graphics::{Direct3D12, Dxgi}, + System::Threading, + }, +}; + +use super::{conv, descriptor, D3D12Lib}; +use crate::{ + auxil::{ + self, + dxgi::{name::ObjectExt as _, result::HResult as _}, + }, + dx12::{ + borrow_optional_interface_temporarily, pipeline_desc::RenderPipelineStateStreamDesc, + shader_compilation, suballocation, DCompLib, DynamicStorageBufferOffsets, Event, + ShaderCacheKey, ShaderCacheValue, + }, + AccelerationStructureEntries, TlasInstance, +}; + +// this has to match Naga's HLSL backend, and also needs to be null-terminated +const NAGA_LOCATION_SEMANTIC: &[u8] = c"LOC".to_bytes(); + +impl super::Device { + #[allow(clippy::too_many_arguments)] + pub(super) fn new( + adapter: auxil::dxgi::factory::DxgiAdapter, + raw: Direct3D12::ID3D12Device, + present_queue: Direct3D12::ID3D12CommandQueue, + features: wgt::Features, + limits: &wgt::Limits, + memory_hints: &wgt::MemoryHints, + private_caps: super::PrivateCapabilities, + library: &Arc, + dcomp_lib: &Arc, + memory_budget_thresholds: wgt::MemoryBudgetThresholds, + compiler_container: Arc, + backend_options: wgt::Dx12BackendOptions, + ) -> Result { + if private_caps + .instance_flags + .contains(wgt::InstanceFlags::VALIDATION) + { + auxil::dxgi::exception::register_exception_handler(); + } + + let mem_allocator = + suballocation::Allocator::new(&raw, memory_hints, memory_budget_thresholds)?; + + let idle_fence: Direct3D12::ID3D12Fence = unsafe { + profiling::scope!("ID3D12Device::CreateFence"); + raw.CreateFence(0, Direct3D12::D3D12_FENCE_FLAG_NONE) + } + .into_device_result("Idle fence creation")?; + + let raw_desc = Direct3D12::D3D12_RESOURCE_DESC { + Dimension: Direct3D12::D3D12_RESOURCE_DIMENSION_BUFFER, + Alignment: 0, + Width: super::ZERO_BUFFER_SIZE, + Height: 1, + DepthOrArraySize: 1, + MipLevels: 1, + Format: Dxgi::Common::DXGI_FORMAT_UNKNOWN, + SampleDesc: Dxgi::Common::DXGI_SAMPLE_DESC { + Count: 1, + Quality: 0, + }, + Layout: Direct3D12::D3D12_TEXTURE_LAYOUT_ROW_MAJOR, + Flags: Direct3D12::D3D12_RESOURCE_FLAG_NONE, + }; + + let heap_properties = Direct3D12::D3D12_HEAP_PROPERTIES { + Type: Direct3D12::D3D12_HEAP_TYPE_CUSTOM, + CPUPageProperty: Direct3D12::D3D12_CPU_PAGE_PROPERTY_NOT_AVAILABLE, + MemoryPoolPreference: match private_caps.memory_architecture { + super::MemoryArchitecture::Unified { .. } => Direct3D12::D3D12_MEMORY_POOL_L0, + super::MemoryArchitecture::NonUnified => Direct3D12::D3D12_MEMORY_POOL_L1, + }, + CreationNodeMask: 0, + VisibleNodeMask: 0, + }; + + profiling::scope!("Zero Buffer Allocation"); + let mut zero_buffer = None::; + unsafe { + raw.CreateCommittedResource( + &heap_properties, + Direct3D12::D3D12_HEAP_FLAG_NONE, + &raw_desc, + Direct3D12::D3D12_RESOURCE_STATE_COMMON, + None, + &mut zero_buffer, + ) + } + .into_device_result("Zero buffer creation")?; + + let zero_buffer = zero_buffer.ok_or(crate::DeviceError::Unexpected)?; + + // Note: without `D3D12_HEAP_FLAG_CREATE_NOT_ZEROED` + // this resource is zeroed by default. + + // maximum number of CBV/SRV/UAV descriptors in heap for Tier 1 + let capacity_views = limits.max_non_sampler_bindings as u64; + + let draw_mesh = if features + .features_wgpu + .contains(wgt::FeaturesWGPU::EXPERIMENTAL_MESH_SHADER) + { + Some(Self::create_command_signature( + &raw, + None, + size_of::(), + &[Direct3D12::D3D12_INDIRECT_ARGUMENT_DESC { + Type: Direct3D12::D3D12_INDIRECT_ARGUMENT_TYPE_DISPATCH_MESH, + ..Default::default() + }], + 0, + )?) + } else { + None + }; + + let shared = super::DeviceShared { + adapter, + zero_buffer, + cmd_signatures: super::CommandSignatures { + draw: Self::create_command_signature( + &raw, + None, + size_of::(), + &[Direct3D12::D3D12_INDIRECT_ARGUMENT_DESC { + Type: Direct3D12::D3D12_INDIRECT_ARGUMENT_TYPE_DRAW, + ..Default::default() + }], + 0, + )?, + draw_indexed: Self::create_command_signature( + &raw, + None, + size_of::(), + &[Direct3D12::D3D12_INDIRECT_ARGUMENT_DESC { + Type: Direct3D12::D3D12_INDIRECT_ARGUMENT_TYPE_DRAW_INDEXED, + ..Default::default() + }], + 0, + )?, + draw_mesh, + dispatch: Self::create_command_signature( + &raw, + None, + size_of::(), + &[Direct3D12::D3D12_INDIRECT_ARGUMENT_DESC { + Type: Direct3D12::D3D12_INDIRECT_ARGUMENT_TYPE_DISPATCH, + ..Default::default() + }], + 0, + )?, + }, + heap_views: descriptor::GeneralHeap::new( + &raw, + Direct3D12::D3D12_DESCRIPTOR_HEAP_TYPE_CBV_SRV_UAV, + capacity_views, + )?, + sampler_heap: super::sampler::SamplerHeap::new(&raw, &private_caps)?, + private_caps, + }; + + let mut rtv_pool = + descriptor::CpuPool::new(raw.clone(), Direct3D12::D3D12_DESCRIPTOR_HEAP_TYPE_RTV); + let null_rtv_handle = rtv_pool.alloc_handle()?; + // A null pResource is used to initialize a null descriptor, + // which guarantees D3D11-like null binding behavior (reading 0s, writes are discarded) + unsafe { + raw.CreateRenderTargetView( + None, + Some(&Direct3D12::D3D12_RENDER_TARGET_VIEW_DESC { + Format: Dxgi::Common::DXGI_FORMAT_R8G8B8A8_UNORM, + ViewDimension: Direct3D12::D3D12_RTV_DIMENSION_TEXTURE2D, + Anonymous: Direct3D12::D3D12_RENDER_TARGET_VIEW_DESC_0 { + Texture2D: Direct3D12::D3D12_TEX2D_RTV { + MipSlice: 0, + PlaneSlice: 0, + }, + }, + }), + null_rtv_handle.raw, + ) + }; + + Ok(super::Device { + raw: raw.clone(), + present_queue, + idler: super::Idler { fence: idle_fence }, + features, + shared: Arc::new(shared), + rtv_pool: Arc::new(Mutex::new(rtv_pool)), + dsv_pool: Mutex::new(descriptor::CpuPool::new( + raw.clone(), + Direct3D12::D3D12_DESCRIPTOR_HEAP_TYPE_DSV, + )), + srv_uav_pool: Mutex::new(descriptor::CpuPool::new( + raw.clone(), + Direct3D12::D3D12_DESCRIPTOR_HEAP_TYPE_CBV_SRV_UAV, + )), + options: backend_options, + library: Arc::clone(library), + dcomp_lib: Arc::clone(dcomp_lib), + #[cfg(feature = "renderdoc")] + render_doc: Default::default(), + null_rtv_handle, + mem_allocator, + compiler_container, + shader_cache: Default::default(), + counters: Default::default(), + }) + } + + fn create_command_signature( + raw: &Direct3D12::ID3D12Device, + root_signature: Option<&Direct3D12::ID3D12RootSignature>, + byte_stride: usize, + arguments: &[Direct3D12::D3D12_INDIRECT_ARGUMENT_DESC], + node_mask: u32, + ) -> Result { + let mut signature = None; + unsafe { + raw.CreateCommandSignature( + &Direct3D12::D3D12_COMMAND_SIGNATURE_DESC { + ByteStride: byte_stride as u32, + NumArgumentDescs: arguments.len() as u32, + pArgumentDescs: arguments.as_ptr(), + NodeMask: node_mask, + }, + root_signature, + &mut signature, + ) + } + .into_device_result("Command signature creation")?; + signature.ok_or(crate::DeviceError::Unexpected) + } + + // Blocks until the dedicated present queue is finished with all of its work. + // + // Once this method completes, the surface is able to be resized or deleted. + pub(super) unsafe fn wait_for_present_queue_idle(&self) -> Result<(), crate::DeviceError> { + let cur_value = unsafe { self.idler.fence.GetCompletedValue() }; + if cur_value == !0 { + return Err(crate::DeviceError::Lost); + } + + let event = Event::create(false, false)?; + + let value = cur_value + 1; + unsafe { self.present_queue.Signal(&self.idler.fence, value) } + .into_device_result("Signal")?; + let hr = unsafe { self.idler.fence.SetEventOnCompletion(value, event.0) }; + hr.into_device_result("Set event")?; + unsafe { Threading::WaitForSingleObject(event.0, Threading::INFINITE) }; + Ok(()) + } + + /// When generating the vertex shader, the fragment stage must be passed if it exists! + /// Otherwise, the generated HLSL may be incorrect since the fragment shader inputs are + /// allowed to be a subset of the vertex outputs. + fn load_shader( + &self, + stage: &crate::ProgrammableStage, + layout: &super::PipelineLayout, + naga_stage: naga::ShaderStage, + fragment_stage: Option<&crate::ProgrammableStage>, + ) -> Result { + let stage_bit = auxil::map_naga_stage(naga_stage); + + let needs_temp_options = stage.zero_initialize_workgroup_memory + != layout.naga_options.zero_initialize_workgroup_memory + || stage.module.runtime_checks.bounds_checks != layout.naga_options.restrict_indexing + || stage.module.runtime_checks.force_loop_bounding + != layout.naga_options.force_loop_bounding + || stage + .module + .runtime_checks + .ray_query_initialization_tracking + != layout.naga_options.ray_query_initialization_tracking; + let mut temp_options; + let naga_options = if needs_temp_options { + temp_options = layout.naga_options.clone(); + temp_options.zero_initialize_workgroup_memory = stage.zero_initialize_workgroup_memory; + temp_options.restrict_indexing = stage.module.runtime_checks.bounds_checks; + temp_options.force_loop_bounding = stage.module.runtime_checks.force_loop_bounding; + temp_options.ray_query_initialization_tracking = stage + .module + .runtime_checks + .ray_query_initialization_tracking; + &temp_options + } else { + &layout.naga_options + }; + + let key = match &stage.module.source { + super::ShaderModuleSource::Naga(naga_shader) => { + use naga::back::hlsl; + + let frag_ep = match fragment_stage { + Some(crate::ProgrammableStage { + module: + super::ShaderModule { + source: super::ShaderModuleSource::Naga(naga_shader), + .. + }, + entry_point, + .. + }) => Some( + hlsl::FragmentEntryPoint::new(&naga_shader.module, entry_point).ok_or( + crate::PipelineError::EntryPoint(naga::ShaderStage::Fragment), + ), + ), + _ => None, + } + .transpose()?; + let (module, info) = naga::back::pipeline_constants::process_overrides( + &naga_shader.module, + &naga_shader.info, + Some((naga_stage, stage.entry_point)), + stage.constants, + ) + .map_err(|e| { + crate::PipelineError::PipelineConstants(stage_bit, format!("HLSL: {e:?}")) + })?; + + let pipeline_options = hlsl::PipelineOptions { + entry_point: Some((naga_stage, stage.entry_point.to_string())), + }; + + //TODO: reuse the writer + let (source, entry_point) = { + let mut source = String::new(); + let mut writer = + hlsl::Writer::new(&mut source, naga_options, &pipeline_options); + + profiling::scope!("naga::back::hlsl::write"); + let mut reflection_info = writer + .write(&module, &info, frag_ep.as_ref()) + .map_err(|e| { + crate::PipelineError::Linkage(stage_bit, format!("HLSL: {e:?}")) + })?; + + assert_eq!(reflection_info.entry_point_names.len(), 1); + + let entry_point = reflection_info + .entry_point_names + .pop() + .unwrap() + .map_err(|e| crate::PipelineError::Linkage(stage_bit, format!("{e}")))?; + + (source, entry_point) + }; + log::debug!( + "Naga generated shader for {entry_point:?} at {naga_stage:?}:\n{source}" + ); + + ShaderCacheKey { + source, + entry_point, + stage: naga_stage, + shader_model: naga_options.shader_model, + } + } + super::ShaderModuleSource::HlslPassthrough(passthrough) => ShaderCacheKey { + source: passthrough.shader.clone(), + entry_point: stage.entry_point.to_string(), + stage: naga_stage, + shader_model: naga_options.shader_model, + }, + super::ShaderModuleSource::DxilPassthrough(passthrough) => { + return Ok(super::CompiledShader::Precompiled( + passthrough.shader.clone(), + )) + } + }; + + { + let mut shader_cache = self.shader_cache.lock(); + let nr_of_shaders_compiled = shader_cache.nr_of_shaders_compiled; + if let Some(value) = shader_cache.entries.get_mut(&key) { + value.last_used = nr_of_shaders_compiled; + return Ok(value.shader.clone()); + } + } + + let source_name = stage.module.raw_name.as_deref(); + + let full_stage = format!("{}_{}", naga_stage.to_hlsl_str(), key.shader_model.to_str()); + + let compiled_shader = self.compiler_container.compile( + self, + &key.source, + source_name, + &key.entry_point, + stage_bit, + &full_stage, + )?; + + { + let mut shader_cache = self.shader_cache.lock(); + shader_cache.nr_of_shaders_compiled += 1; + let nr_of_shaders_compiled = shader_cache.nr_of_shaders_compiled; + let value = ShaderCacheValue { + last_used: nr_of_shaders_compiled, + shader: compiled_shader.clone(), + }; + shader_cache.entries.insert(key, value); + + // Retain all entries that have been used since we compiled the last 100 shaders. + if shader_cache.entries.len() > 200 { + shader_cache + .entries + .retain(|_, v| v.last_used >= nr_of_shaders_compiled - 100); + } + } + + Ok(compiled_shader) + } + + pub fn raw_device(&self) -> &Direct3D12::ID3D12Device { + &self.raw + } + + pub fn raw_queue(&self) -> &Direct3D12::ID3D12CommandQueue { + &self.present_queue + } + + pub unsafe fn texture_from_raw( + resource: Direct3D12::ID3D12Resource, + format: wgt::TextureFormat, + dimension: wgt::TextureDimension, + size: wgt::Extent3d, + mip_level_count: u32, + sample_count: u32, + ) -> super::Texture { + super::Texture { + resource, + format, + dimension, + size, + mip_level_count, + sample_count, + allocation: suballocation::Allocation::none( + suballocation::AllocationType::Texture, + format.theoretical_memory_footprint(size), + ), + } + } + + pub unsafe fn buffer_from_raw( + resource: Direct3D12::ID3D12Resource, + size: wgt::BufferAddress, + ) -> super::Buffer { + super::Buffer { + resource, + size, + allocation: suballocation::Allocation::none( + suballocation::AllocationType::Buffer, + size, + ), + } + } +} + +impl crate::Device for super::Device { + type A = super::Api; + + unsafe fn create_buffer( + &self, + desc: &crate::BufferDescriptor, + ) -> Result { + let mut desc = desc.clone(); + + if desc.usage.contains(wgt::BufferUses::UNIFORM) { + desc.size = desc + .size + .next_multiple_of(Direct3D12::D3D12_CONSTANT_BUFFER_DATA_PLACEMENT_ALIGNMENT.into()) + } + + let (resource, allocation) = + suballocation::DeviceAllocationContext::from(self).create_buffer(&desc)?; + + self.counters.buffers.add(1); + + Ok(super::Buffer { + resource, + size: desc.size, + allocation, + }) + } + + unsafe fn destroy_buffer(&self, buffer: super::Buffer) { + suballocation::DeviceAllocationContext::from(self) + .free_resource(buffer.resource, buffer.allocation); + + self.counters.buffers.sub(1); + } + + unsafe fn add_raw_buffer(&self, _buffer: &super::Buffer) { + self.counters.buffers.add(1); + } + + unsafe fn map_buffer( + &self, + buffer: &super::Buffer, + range: crate::MemoryRange, + ) -> Result { + let mut ptr = ptr::null_mut(); + // TODO: 0 for subresource should be fine here until map and unmap buffer is subresource aware? + unsafe { buffer.resource.Map(0, None, Some(&mut ptr)) }.into_device_result("Map buffer")?; + + Ok(crate::BufferMapping { + ptr: ptr::NonNull::new(unsafe { ptr.offset(range.start as isize).cast::() }) + .unwrap(), + //TODO: double-check this. Documentation is a bit misleading - + // it implies that Map/Unmap is needed to invalidate/flush memory. + is_coherent: true, + }) + } + + unsafe fn unmap_buffer(&self, buffer: &super::Buffer) { + unsafe { buffer.resource.Unmap(0, None) }; + } + + unsafe fn flush_mapped_ranges(&self, _buffer: &super::Buffer, _ranges: I) {} + unsafe fn invalidate_mapped_ranges(&self, _buffer: &super::Buffer, _ranges: I) {} + + unsafe fn create_texture( + &self, + desc: &crate::TextureDescriptor, + ) -> Result { + let raw_desc = Direct3D12::D3D12_RESOURCE_DESC { + Dimension: conv::map_texture_dimension(desc.dimension), + Alignment: 0, + Width: desc.size.width as u64, + Height: desc.size.height, + DepthOrArraySize: desc.size.depth_or_array_layers as u16, + MipLevels: desc.mip_level_count as u16, + Format: auxil::dxgi::conv::map_texture_format_for_resource( + desc.format, + desc.usage, + !desc.view_formats.is_empty(), + self.shared + .private_caps + .casting_fully_typed_format_supported, + ), + SampleDesc: Dxgi::Common::DXGI_SAMPLE_DESC { + Count: desc.sample_count, + Quality: 0, + }, + Layout: Direct3D12::D3D12_TEXTURE_LAYOUT_UNKNOWN, + Flags: conv::map_texture_usage_to_resource_flags(desc.usage), + }; + + let (resource, allocation) = + suballocation::DeviceAllocationContext::from(self).create_texture(desc, raw_desc)?; + + self.counters.textures.add(1); + + Ok(super::Texture { + resource, + format: desc.format, + dimension: desc.dimension, + size: desc.size, + mip_level_count: desc.mip_level_count, + sample_count: desc.sample_count, + allocation, + }) + } + + unsafe fn destroy_texture(&self, texture: super::Texture) { + suballocation::DeviceAllocationContext::from(self) + .free_resource(texture.resource, texture.allocation); + + self.counters.textures.sub(1); + } + + unsafe fn add_raw_texture(&self, _texture: &super::Texture) { + self.counters.textures.add(1); + } + + unsafe fn create_texture_view( + &self, + texture: &super::Texture, + desc: &crate::TextureViewDescriptor, + ) -> Result { + let view_desc = desc.to_internal(texture); + + self.counters.texture_views.add(1); + + Ok(super::TextureView { + raw_format: view_desc.rtv_dsv_format, + aspects: view_desc.aspects, + dimension: desc.dimension, + texture: texture.resource.clone(), + subresource_index: texture.calc_subresource( + desc.range.base_mip_level, + desc.range.base_array_layer, + 0, + ), + mip_slice: desc.range.base_mip_level, + handle_srv: if desc.usage.intersects(wgt::TextureUses::RESOURCE) { + match unsafe { view_desc.to_srv() } { + Some(raw_desc) => { + let handle = self.srv_uav_pool.lock().alloc_handle()?; + unsafe { + self.raw.CreateShaderResourceView( + &texture.resource, + Some(&raw_desc), + handle.raw, + ) + }; + Some(handle) + } + None => None, + } + } else { + None + }, + handle_uav: if desc.usage.intersects( + wgt::TextureUses::STORAGE_READ_ONLY + | wgt::TextureUses::STORAGE_WRITE_ONLY + | wgt::TextureUses::STORAGE_READ_WRITE, + ) { + match unsafe { view_desc.to_uav() } { + Some(raw_desc) => { + let handle = self.srv_uav_pool.lock().alloc_handle()?; + unsafe { + self.raw.CreateUnorderedAccessView( + &texture.resource, + None, + Some(&raw_desc), + handle.raw, + ); + } + Some(handle) + } + None => None, + } + } else { + None + }, + handle_rtv: if desc.usage.intersects(wgt::TextureUses::COLOR_TARGET) + && desc.dimension != wgt::TextureViewDimension::D3 + // 3D RTVs must be created in the render pass + { + let raw_desc = unsafe { view_desc.to_rtv() }; + let handle = self.rtv_pool.lock().alloc_handle()?; + unsafe { + self.raw + .CreateRenderTargetView(&texture.resource, Some(&raw_desc), handle.raw) + }; + Some(handle) + } else { + None + }, + handle_dsv_ro: if desc.usage.intersects(wgt::TextureUses::DEPTH_STENCIL_READ) { + let raw_desc = unsafe { view_desc.to_dsv(true) }; + let handle = self.dsv_pool.lock().alloc_handle()?; + unsafe { + self.raw + .CreateDepthStencilView(&texture.resource, Some(&raw_desc), handle.raw) + }; + Some(handle) + } else { + None + }, + handle_dsv_rw: if desc.usage.intersects(wgt::TextureUses::DEPTH_STENCIL_WRITE) { + let raw_desc = unsafe { view_desc.to_dsv(false) }; + let handle = self.dsv_pool.lock().alloc_handle()?; + unsafe { + self.raw + .CreateDepthStencilView(&texture.resource, Some(&raw_desc), handle.raw) + }; + Some(handle) + } else { + None + }, + }) + } + + unsafe fn destroy_texture_view(&self, view: super::TextureView) { + if view.handle_srv.is_some() || view.handle_uav.is_some() { + let mut pool = self.srv_uav_pool.lock(); + if let Some(handle) = view.handle_srv { + pool.free_handle(handle); + } + if let Some(handle) = view.handle_uav { + pool.free_handle(handle); + } + } + if let Some(handle) = view.handle_rtv { + self.rtv_pool.lock().free_handle(handle); + } + if view.handle_dsv_ro.is_some() || view.handle_dsv_rw.is_some() { + let mut pool = self.dsv_pool.lock(); + if let Some(handle) = view.handle_dsv_ro { + pool.free_handle(handle); + } + if let Some(handle) = view.handle_dsv_rw { + pool.free_handle(handle); + } + } + + self.counters.texture_views.sub(1); + } + + unsafe fn create_sampler( + &self, + desc: &crate::SamplerDescriptor, + ) -> Result { + let reduction = match desc.compare { + Some(_) => Direct3D12::D3D12_FILTER_REDUCTION_TYPE_COMPARISON, + None => Direct3D12::D3D12_FILTER_REDUCTION_TYPE_STANDARD, + }; + let mut filter = Direct3D12::D3D12_FILTER( + (conv::map_filter_mode(desc.min_filter).0 << Direct3D12::D3D12_MIN_FILTER_SHIFT) + | (conv::map_filter_mode(desc.mag_filter).0 << Direct3D12::D3D12_MAG_FILTER_SHIFT) + | (conv::map_mipmap_filter_mode(desc.mipmap_filter).0 + << Direct3D12::D3D12_MIP_FILTER_SHIFT) + | (reduction.0 << Direct3D12::D3D12_FILTER_REDUCTION_TYPE_SHIFT), + ); + + if desc.anisotropy_clamp != 1 { + filter.0 |= Direct3D12::D3D12_FILTER_ANISOTROPIC.0; + }; + + let border_color = conv::map_border_color(desc.border_color); + + let raw_desc = Direct3D12::D3D12_SAMPLER_DESC { + Filter: filter, + AddressU: conv::map_address_mode(desc.address_modes[0]), + AddressV: conv::map_address_mode(desc.address_modes[1]), + AddressW: conv::map_address_mode(desc.address_modes[2]), + MipLODBias: 0f32, + MaxAnisotropy: desc.anisotropy_clamp as u32, + + ComparisonFunc: conv::map_comparison(desc.compare.unwrap_or_default()), + BorderColor: border_color, + MinLOD: desc.lod_clamp.start, + MaxLOD: desc.lod_clamp.end, + }; + + let index = self + .shared + .sampler_heap + .create_sampler(&self.raw, raw_desc)?; + + self.counters.samplers.add(1); + + Ok(super::Sampler { + index, + desc: raw_desc, + }) + } + + unsafe fn destroy_sampler(&self, sampler: super::Sampler) { + self.shared + .sampler_heap + .destroy_sampler(sampler.desc, sampler.index); + self.counters.samplers.sub(1); + } + + unsafe fn create_command_encoder( + &self, + desc: &crate::CommandEncoderDescriptor, + ) -> Result { + let allocator: Direct3D12::ID3D12CommandAllocator = unsafe { + self.raw + .CreateCommandAllocator(Direct3D12::D3D12_COMMAND_LIST_TYPE_DIRECT) + } + .into_device_result("Command allocator creation")?; + + if let Some(label) = desc.label { + allocator.set_name(label)?; + } + + self.counters.command_encoders.add(1); + + Ok(super::CommandEncoder { + allocator, + device: self.raw.clone(), + shared: Arc::clone(&self.shared), + mem_allocator: self.mem_allocator.clone(), + rtv_pool: Arc::clone(&self.rtv_pool), + temp_rtv_handles: Vec::new(), + intermediate_copy_bufs: Vec::new(), + null_rtv_handle: self.null_rtv_handle, + list: None, + free_lists: Vec::new(), + pass: super::PassState::new(), + temp: super::Temp::default(), + end_of_pass_timer_query: None, + counters: Arc::clone(&self.counters), + }) + } + + unsafe fn create_bind_group_layout( + &self, + desc: &crate::BindGroupLayoutDescriptor, + ) -> Result { + let mut num_views = 0; + let mut has_sampler_in_group = false; + for entry in desc.entries.iter() { + let count = entry.count.map_or(1, NonZeroU32::get); + match entry.ty { + wgt::BindingType::Buffer { + ty: wgt::BufferBindingType::Uniform, + has_dynamic_offset: true, + .. + } => {} + wgt::BindingType::Buffer { .. } + | wgt::BindingType::Texture { .. } + | wgt::BindingType::StorageTexture { .. } + | wgt::BindingType::AccelerationStructure { .. } => num_views += count, + wgt::BindingType::Sampler { .. } => has_sampler_in_group = true, + // Three texture planes and one params buffer + wgt::BindingType::ExternalTexture => num_views += 4 * count, + } + } + + if has_sampler_in_group { + num_views += 1; + } + + self.counters.bind_group_layouts.add(1); + + Ok(super::BindGroupLayout { + entries: desc.entries.to_vec(), + cpu_heap_views: if num_views != 0 { + let heap = descriptor::CpuHeap::new( + &self.raw, + Direct3D12::D3D12_DESCRIPTOR_HEAP_TYPE_CBV_SRV_UAV, + num_views, + )?; + Some(heap) + } else { + None + }, + copy_counts: vec![1; num_views as usize], + }) + } + + unsafe fn destroy_bind_group_layout(&self, _bg_layout: super::BindGroupLayout) { + self.counters.bind_group_layouts.sub(1); + } + + unsafe fn create_pipeline_layout( + &self, + desc: &crate::PipelineLayoutDescriptor, + ) -> Result { + use naga::back::hlsl; + // Pipeline layouts are implemented as RootSignature for D3D12. + // + // Immediates are implemented as root constants. + // + // Each bind group layout might use one SRV/CBV/UAV descriptor table. + // With resources in the bind group layout using: + // - 1 CBV per non-dynamic uniform buffer + // - 1 SRV per acceleration structure + // - 1 SRV for all samplers in a bind group + // - 1 SRV per texture + // - 1 SRV per read-only storage buffer + // - 1 UAV per storage texture + // - 1 UAV per read-write storage buffer + // - 3 SRVs & 1 CBV per external texture + // + // Each dynamic uniform buffer takes up a CBV root descriptor. + // This is easier than trying to patch up the offset on the shader side. + // + // Each dynamic storage buffer is an SRV or UAV in the descriptor table + // and its dynamic offsets are passed via root constants. + // + // All samplers go into a single sampler descriptor table. + // + // 3 additional root constants are used to populate built-in (shader) inputs. + // + // Root signature layout: + // Root Constants: Parameter=0, Space=0 + // ... + // (bind group [0]) - Space=0 + // View descriptor table, if any + // Sampler buffer descriptor table, if any + // Root descriptors (for dynamic offset buffers) + // (bind group [1]) - Space=0 + // ... + // (bind group [2]) - Space=0 + // Special constant buffer: Space=0 + // Sampler descriptor tables: Space=0 + // SamplerState Array: Space=0, Register=0-2047 + // SamplerComparisonState Array: Space=0, Register=2048-4095 + + //TODO: put lower bind group indices further down the root signature. See: + // https://microsoft.github.io/DirectX-Specs/d3d/ResourceBinding.html#binding-model + // Currently impossible because wgpu-core only re-binds the descriptor sets based + // on Vulkan-like layout compatibility rules. + + let mut binding_map = hlsl::BindingMap::default(); + let mut sampler_buffer_binding_map = hlsl::SamplerIndexBufferBindingMap::default(); + let mut external_texture_binding_map = hlsl::ExternalTextureBindingMap::default(); + let mut bind_cbv = hlsl::BindTarget::default(); + let mut bind_srv = hlsl::BindTarget::default(); + let mut bind_uav = hlsl::BindTarget::default(); + let mut parameters = Vec::new(); + let mut immediates_target = None; + let mut root_constant_info = None; + + if desc.immediate_size != 0 { + let parameter_index = parameters.len(); + let size = desc.immediate_size / 4; + parameters.push(Direct3D12::D3D12_ROOT_PARAMETER { + ParameterType: Direct3D12::D3D12_ROOT_PARAMETER_TYPE_32BIT_CONSTANTS, + Anonymous: Direct3D12::D3D12_ROOT_PARAMETER_0 { + Constants: Direct3D12::D3D12_ROOT_CONSTANTS { + ShaderRegister: bind_cbv.register, + RegisterSpace: bind_cbv.space as u32, + Num32BitValues: size, + }, + }, + ShaderVisibility: Direct3D12::D3D12_SHADER_VISIBILITY_ALL, + }); + let binding = bind_cbv; + bind_cbv.register += 1; + root_constant_info = Some(super::RootConstantInfo { + root_index: parameter_index as u32, + range: 0..size, + }); + immediates_target = Some(binding); + + bind_cbv.space += 1; + } + + let mut dynamic_storage_buffer_offsets_targets = alloc::collections::BTreeMap::new(); + let mut total_dynamic_storage_buffers = 0; + + // Collect the whole number of bindings we will create upfront. + // It allows us to preallocate enough storage to avoid reallocation, + // which could cause invalid pointers. + let mut total_non_dynamic_entries = 0_usize; + let mut sampler_in_any_bind_group = false; + for bgl in desc.bind_group_layouts { + let Some(bgl) = bgl else { + continue; + }; + + let mut sampler_in_bind_group = false; + + for entry in &bgl.entries { + match entry.ty { + wgt::BindingType::Buffer { + ty: wgt::BufferBindingType::Uniform, + has_dynamic_offset: true, + .. + } => {} + wgt::BindingType::Sampler(_) => sampler_in_bind_group = true, + // Three texture planes and one params buffer + wgt::BindingType::ExternalTexture => total_non_dynamic_entries += 4, + _ => total_non_dynamic_entries += 1, + } + } + + if sampler_in_bind_group { + // One for the sampler buffer + total_non_dynamic_entries += 1; + sampler_in_any_bind_group = true; + } + } + + if sampler_in_any_bind_group { + // Two for the sampler arrays themselves + total_non_dynamic_entries += 2; + } + + let mut ranges = Vec::with_capacity(total_non_dynamic_entries); + + let mut bind_group_infos = [const { None }; crate::MAX_BIND_GROUPS]; + for (index, bgl) in desc.bind_group_layouts.iter().enumerate() { + let Some(bgl) = bgl else { + continue; + }; + + let mut info = super::BindGroupInfo { + tables: super::TableTypes::empty(), + base_root_index: parameters.len() as u32, + dynamic_storage_buffer_offsets: None, + }; + + let mut visibility_view_static = wgt::ShaderStages::empty(); + let mut visibility_view_dynamic_uniform = wgt::ShaderStages::empty(); + let mut visibility_view_dynamic_storage = wgt::ShaderStages::empty(); + for entry in bgl.entries.iter() { + match entry.ty { + wgt::BindingType::Sampler { .. } => { + visibility_view_static |= wgt::ShaderStages::all() + } + wgt::BindingType::Buffer { + ty: wgt::BufferBindingType::Uniform, + has_dynamic_offset: true, + .. + } => visibility_view_dynamic_uniform |= entry.visibility, + wgt::BindingType::Buffer { + ty: wgt::BufferBindingType::Storage { .. }, + has_dynamic_offset: true, + .. + } => visibility_view_dynamic_storage |= entry.visibility, + _ => visibility_view_static |= entry.visibility, + } + } + + let mut dynamic_storage_buffers = 0; + + // SRV/CBV/UAV descriptor tables + let range_base = ranges.len(); + for entry in bgl.entries.iter() { + let count = entry.count.map_or(1, NonZeroU32::get); + if let wgt::BindingType::ExternalTexture = entry.ty { + // External textures need 3 SRVs (a texture for each plane) + // and 1 CBV for the parameters buffer. + let bind_target = hlsl::ExternalTextureBindTarget { + planes: core::array::from_fn(|_| hlsl::BindTarget { + register: { + let register = bind_srv.register; + bind_srv.register += count; + register + }, + ..bind_srv + }), + params: hlsl::BindTarget { + register: { + let register = bind_cbv.register; + bind_cbv.register += count; + register + }, + ..bind_cbv + }, + }; + external_texture_binding_map.insert( + naga::ResourceBinding { + group: index as u32, + binding: entry.binding, + }, + bind_target, + ); + for bt in bind_target.planes { + ranges.push(Direct3D12::D3D12_DESCRIPTOR_RANGE { + RangeType: Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE_SRV, + NumDescriptors: count, + BaseShaderRegister: bt.register, + RegisterSpace: bt.space as u32, + OffsetInDescriptorsFromTableStart: + Direct3D12::D3D12_DESCRIPTOR_RANGE_OFFSET_APPEND, + }); + } + ranges.push(Direct3D12::D3D12_DESCRIPTOR_RANGE { + RangeType: Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE_CBV, + NumDescriptors: count, + BaseShaderRegister: bind_target.params.register, + RegisterSpace: bind_target.params.space as u32, + OffsetInDescriptorsFromTableStart: + Direct3D12::D3D12_DESCRIPTOR_RANGE_OFFSET_APPEND, + }); + } else { + let (range_ty, has_dynamic_offset) = match entry.ty { + wgt::BindingType::Buffer { + ty, + has_dynamic_offset: true, + .. + } => match ty { + wgt::BufferBindingType::Uniform => continue, + wgt::BufferBindingType::Storage { .. } => { + (conv::map_binding_type(&entry.ty), true) + } + }, + ref other => (conv::map_binding_type(other), false), + }; + let bt = match range_ty { + Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE_CBV => &mut bind_cbv, + Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE_SRV => &mut bind_srv, + Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE_UAV => &mut bind_uav, + Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE_SAMPLER => continue, + _ => todo!(), + }; + + let binding_array_size = entry.count.map(NonZeroU32::get); + + let dynamic_storage_buffer_offsets_index = if has_dynamic_offset { + debug_assert!( + binding_array_size.is_none(), + "binding arrays and dynamic buffers are mutually exclusive" + ); + let ret = Some(dynamic_storage_buffers); + dynamic_storage_buffers += 1; + ret + } else { + None + }; + + binding_map.insert( + naga::ResourceBinding { + group: index as u32, + binding: entry.binding, + }, + hlsl::BindTarget { + binding_array_size, + dynamic_storage_buffer_offsets_index, + ..*bt + }, + ); + ranges.push(Direct3D12::D3D12_DESCRIPTOR_RANGE { + RangeType: range_ty, + NumDescriptors: count, + BaseShaderRegister: bt.register, + RegisterSpace: bt.space as u32, + OffsetInDescriptorsFromTableStart: + Direct3D12::D3D12_DESCRIPTOR_RANGE_OFFSET_APPEND, + }); + bt.register += count; + } + } + + let mut sampler_index_within_bind_group = 0; + for entry in bgl.entries.iter() { + if let wgt::BindingType::Sampler(_) = entry.ty { + binding_map.insert( + naga::ResourceBinding { + group: index as u32, + binding: entry.binding, + }, + hlsl::BindTarget { + // Naga does not use the space field for samplers + space: 255, + register: sampler_index_within_bind_group, + binding_array_size: None, + dynamic_storage_buffer_offsets_index: None, + restrict_indexing: false, + }, + ); + sampler_index_within_bind_group += 1; + } + } + + if sampler_index_within_bind_group != 0 { + sampler_buffer_binding_map.insert( + hlsl::SamplerIndexBufferKey { + group: index as u32, + }, + bind_srv, + ); + ranges.push(Direct3D12::D3D12_DESCRIPTOR_RANGE { + RangeType: Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE_SRV, + NumDescriptors: 1, + BaseShaderRegister: bind_srv.register, + RegisterSpace: bind_srv.space as u32, + OffsetInDescriptorsFromTableStart: + Direct3D12::D3D12_DESCRIPTOR_RANGE_OFFSET_APPEND, + }); + bind_srv.register += 1; + } + + if ranges.len() > range_base { + let range = &ranges[range_base..]; + parameters.push(Direct3D12::D3D12_ROOT_PARAMETER { + ParameterType: Direct3D12::D3D12_ROOT_PARAMETER_TYPE_DESCRIPTOR_TABLE, + Anonymous: Direct3D12::D3D12_ROOT_PARAMETER_0 { + DescriptorTable: Direct3D12::D3D12_ROOT_DESCRIPTOR_TABLE { + NumDescriptorRanges: range.len() as u32, + pDescriptorRanges: range.as_ptr(), + }, + }, + ShaderVisibility: conv::map_visibility(visibility_view_static), + }); + info.tables |= super::TableTypes::SRV_CBV_UAV; + } + + // Root descriptors for dynamic uniform buffers + let dynamic_buffers_visibility = conv::map_visibility(visibility_view_dynamic_uniform); + for entry in bgl.entries.iter() { + match entry.ty { + wgt::BindingType::Buffer { + ty: wgt::BufferBindingType::Uniform, + has_dynamic_offset: true, + .. + } => {} + _ => continue, + }; + + binding_map.insert( + naga::ResourceBinding { + group: index as u32, + binding: entry.binding, + }, + hlsl::BindTarget { + binding_array_size: entry.count.map(NonZeroU32::get), + restrict_indexing: true, + ..bind_cbv + }, + ); + + parameters.push(Direct3D12::D3D12_ROOT_PARAMETER { + ParameterType: Direct3D12::D3D12_ROOT_PARAMETER_TYPE_CBV, + Anonymous: Direct3D12::D3D12_ROOT_PARAMETER_0 { + Descriptor: Direct3D12::D3D12_ROOT_DESCRIPTOR { + ShaderRegister: bind_cbv.register, + RegisterSpace: bind_cbv.space as u32, + }, + }, + ShaderVisibility: dynamic_buffers_visibility, + }); + + bind_cbv.register += entry.count.map_or(1, NonZeroU32::get); + } + + // Root constants for (offsets of) dynamic storage buffers + if dynamic_storage_buffers > 0 { + let parameter_index = parameters.len(); + + parameters.push(Direct3D12::D3D12_ROOT_PARAMETER { + ParameterType: Direct3D12::D3D12_ROOT_PARAMETER_TYPE_32BIT_CONSTANTS, + Anonymous: Direct3D12::D3D12_ROOT_PARAMETER_0 { + Constants: Direct3D12::D3D12_ROOT_CONSTANTS { + ShaderRegister: bind_cbv.register, + RegisterSpace: bind_cbv.space as u32, + Num32BitValues: dynamic_storage_buffers, + }, + }, + ShaderVisibility: conv::map_visibility(visibility_view_dynamic_storage), + }); + + let binding = hlsl::OffsetsBindTarget { + space: bind_cbv.space, + register: bind_cbv.register, + size: dynamic_storage_buffers, + }; + + bind_cbv.register += 1; + + dynamic_storage_buffer_offsets_targets.insert(index as u32, binding); + info.dynamic_storage_buffer_offsets = Some(DynamicStorageBufferOffsets { + root_index: parameter_index as u32, + range: total_dynamic_storage_buffers as usize + ..total_dynamic_storage_buffers as usize + dynamic_storage_buffers as usize, + }); + total_dynamic_storage_buffers += dynamic_storage_buffers; + } + + bind_group_infos[index] = Some(info); + } + + let sampler_heap_target = hlsl::SamplerHeapBindTargets { + standard_samplers: hlsl::BindTarget { + space: 0, + register: 0, + binding_array_size: None, + dynamic_storage_buffer_offsets_index: None, + restrict_indexing: false, + }, + comparison_samplers: hlsl::BindTarget { + space: 0, + register: 2048, + binding_array_size: None, + dynamic_storage_buffer_offsets_index: None, + restrict_indexing: false, + }, + }; + + let mut sampler_heap_root_index = None; + if sampler_in_any_bind_group { + // Sampler descriptor tables + // + // We bind two sampler ranges pointing to the same descriptor heap, using two different register ranges. + // + // We bind them as normal samplers in registers 0-2047 and comparison samplers in registers 2048-4095. + // Tier 2 hardware guarantees that the type of sampler only needs to match if the sampler is actually + // accessed in the shader. As such, we can bind the same array of samplers to both registers. + // + // We do this because HLSL does not allow you to alias registers at all. + let range_base = ranges.len(); + // Standard samplers, registers 0-2047 + ranges.push(Direct3D12::D3D12_DESCRIPTOR_RANGE { + RangeType: Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE_SAMPLER, + NumDescriptors: 2048, + BaseShaderRegister: 0, + RegisterSpace: 0, + OffsetInDescriptorsFromTableStart: 0, + }); + // Comparison samplers, registers 2048-4095 + ranges.push(Direct3D12::D3D12_DESCRIPTOR_RANGE { + RangeType: Direct3D12::D3D12_DESCRIPTOR_RANGE_TYPE_SAMPLER, + NumDescriptors: 2048, + BaseShaderRegister: 2048, + RegisterSpace: 0, + OffsetInDescriptorsFromTableStart: 0, + }); + + let range = &ranges[range_base..]; + sampler_heap_root_index = Some(parameters.len() as super::RootIndex); + parameters.push(Direct3D12::D3D12_ROOT_PARAMETER { + ParameterType: Direct3D12::D3D12_ROOT_PARAMETER_TYPE_DESCRIPTOR_TABLE, + Anonymous: Direct3D12::D3D12_ROOT_PARAMETER_0 { + DescriptorTable: Direct3D12::D3D12_ROOT_DESCRIPTOR_TABLE { + NumDescriptorRanges: range.len() as u32, + pDescriptorRanges: range.as_ptr(), + }, + }, + ShaderVisibility: Direct3D12::D3D12_SHADER_VISIBILITY_ALL, + }); + } + + // Ensure that we didn't reallocate! + debug_assert_eq!(ranges.len(), total_non_dynamic_entries); + + let (special_constants_root_index, special_constants_binding) = if desc.flags.intersects( + crate::PipelineLayoutFlags::FIRST_VERTEX_INSTANCE + | crate::PipelineLayoutFlags::NUM_WORK_GROUPS, + ) { + let parameter_index = parameters.len(); + parameters.push(Direct3D12::D3D12_ROOT_PARAMETER { + ParameterType: Direct3D12::D3D12_ROOT_PARAMETER_TYPE_32BIT_CONSTANTS, + Anonymous: Direct3D12::D3D12_ROOT_PARAMETER_0 { + Constants: Direct3D12::D3D12_ROOT_CONSTANTS { + ShaderRegister: bind_cbv.register, + RegisterSpace: bind_cbv.space as u32, + Num32BitValues: 3, // 0 = first_vertex, 1 = first_instance, 2 = other + }, + }, + ShaderVisibility: Direct3D12::D3D12_SHADER_VISIBILITY_ALL, // really needed for VS and CS only, + }); + let binding = bind_cbv; + // This is the last time we use this, but lets increment + // it so if we add more later, the value behaves correctly. + + // This is an allow as it doesn't trigger on 1.90, hal's MSRV. + #[allow(unused_assignments)] + { + bind_cbv.register += 1; + } + (Some(parameter_index as u32), Some(binding)) + } else { + (None, None) + }; + + let blob = self.library.serialize_root_signature( + Direct3D12::D3D_ROOT_SIGNATURE_VERSION_1_0, + ¶meters, + &[], + Direct3D12::D3D12_ROOT_SIGNATURE_FLAG_ALLOW_INPUT_ASSEMBLER_INPUT_LAYOUT, + )?; + + let raw = unsafe { + self.raw + .CreateRootSignature::(0, blob.as_slice()) + } + .into_device_result("Root signature creation")?; + + let special_constants = if let Some(root_index) = special_constants_root_index { + let cmd_signatures = if desc + .flags + .contains(crate::PipelineLayoutFlags::INDIRECT_BUILTIN_UPDATE) + { + let constant_indirect_argument_desc = Direct3D12::D3D12_INDIRECT_ARGUMENT_DESC { + Type: Direct3D12::D3D12_INDIRECT_ARGUMENT_TYPE_CONSTANT, + Anonymous: Direct3D12::D3D12_INDIRECT_ARGUMENT_DESC_0 { + Constant: Direct3D12::D3D12_INDIRECT_ARGUMENT_DESC_0_1 { + RootParameterIndex: root_index, + DestOffsetIn32BitValues: 0, + Num32BitValuesToSet: 3, + }, + }, + }; + let special_constant_buffer_args_len = { + // Hack: construct a dummy value of the special constants buffer value we need to + // fill, and calculate the size of each member. + let super::RootElement::SpecialConstantBuffer { + first_vertex, + first_instance, + other, + } = (super::RootElement::SpecialConstantBuffer { + first_vertex: 0, + first_instance: 0, + other: 0, + }) + else { + unreachable!(); + }; + size_of_val(&first_vertex) + size_of_val(&first_instance) + size_of_val(&other) + }; + + let draw_mesh = if self + .features + .features_wgpu + .contains(wgt::FeaturesWGPU::EXPERIMENTAL_MESH_SHADER) + { + Some(Self::create_command_signature( + &self.raw, + Some(&raw), + special_constant_buffer_args_len + size_of::(), + &[ + constant_indirect_argument_desc, + Direct3D12::D3D12_INDIRECT_ARGUMENT_DESC { + Type: Direct3D12::D3D12_INDIRECT_ARGUMENT_TYPE_DISPATCH_MESH, + ..Default::default() + }, + ], + 0, + )?) + } else { + None + }; + + Some(super::CommandSignatures { + draw: Self::create_command_signature( + &self.raw, + Some(&raw), + special_constant_buffer_args_len + size_of::(), + &[ + constant_indirect_argument_desc, + Direct3D12::D3D12_INDIRECT_ARGUMENT_DESC { + Type: Direct3D12::D3D12_INDIRECT_ARGUMENT_TYPE_DRAW, + ..Default::default() + }, + ], + 0, + )?, + draw_indexed: Self::create_command_signature( + &self.raw, + Some(&raw), + special_constant_buffer_args_len + + size_of::(), + &[ + constant_indirect_argument_desc, + Direct3D12::D3D12_INDIRECT_ARGUMENT_DESC { + Type: Direct3D12::D3D12_INDIRECT_ARGUMENT_TYPE_DRAW_INDEXED, + ..Default::default() + }, + ], + 0, + )?, + draw_mesh, + dispatch: Self::create_command_signature( + &self.raw, + Some(&raw), + special_constant_buffer_args_len + size_of::(), + &[ + constant_indirect_argument_desc, + Direct3D12::D3D12_INDIRECT_ARGUMENT_DESC { + Type: Direct3D12::D3D12_INDIRECT_ARGUMENT_TYPE_DISPATCH, + ..Default::default() + }, + ], + 0, + )?, + }) + } else { + None + }; + Some(super::PipelineLayoutSpecialConstants { + root_index, + indirect_cmd_signatures: cmd_signatures, + }) + } else { + None + }; + + if let Some(label) = desc.label { + raw.set_name(label)?; + } + + self.counters.pipeline_layouts.add(1); + + Ok(super::PipelineLayout { + shared: super::PipelineLayoutShared { + signature: Some(raw), + total_root_elements: parameters.len() as super::RootIndex, + special_constants, + root_constant_info, + sampler_heap_root_index, + }, + bind_group_infos, + naga_options: hlsl::Options { + shader_model: self.shared.private_caps.shader_model, + binding_map, + fake_missing_bindings: false, + special_constants_binding, + immediates_target, + dynamic_storage_buffer_offsets_targets, + zero_initialize_workgroup_memory: true, + restrict_indexing: true, + sampler_heap_target, + sampler_buffer_binding_map, + external_texture_binding_map, + force_loop_bounding: true, + ray_query_initialization_tracking: true, + }, + }) + } + + unsafe fn destroy_pipeline_layout(&self, _pipeline_layout: super::PipelineLayout) { + self.counters.pipeline_layouts.sub(1); + } + + unsafe fn create_bind_group( + &self, + desc: &crate::BindGroupDescriptor< + super::BindGroupLayout, + super::Buffer, + super::Sampler, + super::TextureView, + super::AccelerationStructure, + >, + ) -> Result { + let mut cpu_views = desc + .layout + .cpu_heap_views + .as_ref() + .map(|cpu_heap| cpu_heap.inner.lock()); + if let Some(ref mut inner) = cpu_views { + inner.stage.clear(); + } + let mut dynamic_buffers = Vec::new(); + + let layout_and_entry_iter = desc.entries.iter().map(|entry| { + let layout = desc + .layout + .entries + .iter() + .find(|layout_entry| layout_entry.binding == entry.binding) + .expect("internal error: no layout entry found with binding slot"); + (layout, entry) + }); + let mut sampler_indexes: Vec = Vec::new(); + + for (layout, entry) in layout_and_entry_iter { + match layout.ty { + wgt::BindingType::Buffer { + ty, + has_dynamic_offset, + .. + } => { + let start = entry.resource_index as usize; + let end = start + entry.count as usize; + for data in &desc.buffers[start..end] { + let gpu_address = data.resolve_address(); + let mut size = data.resolve_size().try_into().unwrap(); + + if has_dynamic_offset { + match ty { + wgt::BufferBindingType::Uniform => { + dynamic_buffers.push(super::DynamicBuffer::Uniform( + Direct3D12::D3D12_GPU_DESCRIPTOR_HANDLE { + ptr: data.resolve_address(), + }, + )); + continue; + } + wgt::BufferBindingType::Storage { .. } => { + size = (data.buffer.size - data.offset) as u32; + dynamic_buffers.push(super::DynamicBuffer::Storage); + } + } + } + + let inner = cpu_views.as_mut().unwrap(); + let cpu_index = inner.stage.len() as u32; + let handle = desc.layout.cpu_heap_views.as_ref().unwrap().at(cpu_index); + match ty { + wgt::BufferBindingType::Uniform => { + let size_mask = + Direct3D12::D3D12_CONSTANT_BUFFER_DATA_PLACEMENT_ALIGNMENT - 1; + let raw_desc = Direct3D12::D3D12_CONSTANT_BUFFER_VIEW_DESC { + BufferLocation: gpu_address, + SizeInBytes: ((size - 1) | size_mask) + 1, + }; + unsafe { + self.raw.CreateConstantBufferView(Some(&raw_desc), handle) + }; + } + wgt::BufferBindingType::Storage { read_only: true } => { + let raw_desc = Direct3D12::D3D12_SHADER_RESOURCE_VIEW_DESC { + Format: Dxgi::Common::DXGI_FORMAT_R32_TYPELESS, + Shader4ComponentMapping: + Direct3D12::D3D12_DEFAULT_SHADER_4_COMPONENT_MAPPING, + ViewDimension: Direct3D12::D3D12_SRV_DIMENSION_BUFFER, + Anonymous: Direct3D12::D3D12_SHADER_RESOURCE_VIEW_DESC_0 { + Buffer: Direct3D12::D3D12_BUFFER_SRV { + FirstElement: data.offset / 4, + NumElements: size / 4, + StructureByteStride: 0, + Flags: Direct3D12::D3D12_BUFFER_SRV_FLAG_RAW, + }, + }, + }; + unsafe { + self.raw.CreateShaderResourceView( + &data.buffer.resource, + Some(&raw_desc), + handle, + ) + }; + } + wgt::BufferBindingType::Storage { read_only: false } => { + let raw_desc = Direct3D12::D3D12_UNORDERED_ACCESS_VIEW_DESC { + Format: Dxgi::Common::DXGI_FORMAT_R32_TYPELESS, + ViewDimension: Direct3D12::D3D12_UAV_DIMENSION_BUFFER, + Anonymous: Direct3D12::D3D12_UNORDERED_ACCESS_VIEW_DESC_0 { + Buffer: Direct3D12::D3D12_BUFFER_UAV { + FirstElement: data.offset / 4, + NumElements: size / 4, + StructureByteStride: 0, + CounterOffsetInBytes: 0, + Flags: Direct3D12::D3D12_BUFFER_UAV_FLAG_RAW, + }, + }, + }; + unsafe { + self.raw.CreateUnorderedAccessView( + &data.buffer.resource, + None, + Some(&raw_desc), + handle, + ) + }; + } + } + inner.stage.push(handle); + } + } + wgt::BindingType::Texture { .. } => { + let start = entry.resource_index as usize; + let end = start + entry.count as usize; + for data in &desc.textures[start..end] { + let handle = data.view.handle_srv.unwrap(); + cpu_views.as_mut().unwrap().stage.push(handle.raw); + } + } + wgt::BindingType::StorageTexture { .. } => { + let start = entry.resource_index as usize; + let end = start + entry.count as usize; + for data in &desc.textures[start..end] { + let handle = data.view.handle_uav.unwrap(); + cpu_views.as_mut().unwrap().stage.push(handle.raw); + } + } + wgt::BindingType::Sampler { .. } => { + let start = entry.resource_index as usize; + let end = start + entry.count as usize; + for &data in &desc.samplers[start..end] { + sampler_indexes.push(data.index); + } + } + wgt::BindingType::AccelerationStructure { .. } => { + let start = entry.resource_index as usize; + let end = start + entry.count as usize; + for data in &desc.acceleration_structures[start..end] { + let inner = cpu_views.as_mut().unwrap(); + let cpu_index = inner.stage.len() as u32; + let handle = desc.layout.cpu_heap_views.as_ref().unwrap().at(cpu_index); + let raw_desc = Direct3D12::D3D12_SHADER_RESOURCE_VIEW_DESC { + Format: Dxgi::Common::DXGI_FORMAT_UNKNOWN, + Shader4ComponentMapping: + Direct3D12::D3D12_DEFAULT_SHADER_4_COMPONENT_MAPPING, + ViewDimension: + Direct3D12::D3D12_SRV_DIMENSION_RAYTRACING_ACCELERATION_STRUCTURE, + Anonymous: Direct3D12::D3D12_SHADER_RESOURCE_VIEW_DESC_0 { + RaytracingAccelerationStructure: + Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_SRV { + Location: unsafe { data.resource.GetGPUVirtualAddress() }, + }, + }, + }; + unsafe { + self.raw + .CreateShaderResourceView(None, Some(&raw_desc), handle) + }; + inner.stage.push(handle); + } + } + wgt::BindingType::ExternalTexture => { + // We don't yet support binding arrays of external textures. + // https://github.com/gfx-rs/wgpu/issues/8027 + assert_eq!(entry.count, 1); + let external_texture = &desc.external_textures[entry.resource_index as usize]; + for plane in &external_texture.planes { + let plane_handle = plane.view.handle_srv.unwrap(); + cpu_views.as_mut().unwrap().stage.push(plane_handle.raw); + } + let gpu_address = external_texture.params.resolve_address(); + let size = external_texture.params.resolve_size() as u32; + let inner = cpu_views.as_mut().unwrap(); + let cpu_index = inner.stage.len() as u32; + let params_handle = desc.layout.cpu_heap_views.as_ref().unwrap().at(cpu_index); + let size_mask = Direct3D12::D3D12_CONSTANT_BUFFER_DATA_PLACEMENT_ALIGNMENT - 1; + let raw_desc = Direct3D12::D3D12_CONSTANT_BUFFER_VIEW_DESC { + BufferLocation: gpu_address, + SizeInBytes: ((size - 1) | size_mask) + 1, + }; + unsafe { + self.raw + .CreateConstantBufferView(Some(&raw_desc), params_handle) + }; + inner.stage.push(params_handle); + } + } + } + + let sampler_index_buffer = if !sampler_indexes.is_empty() { + let buffer_size = (sampler_indexes.len() * size_of::()) as u64; + + let label = if let Some(label) = desc.label { + Cow::Owned(format!("{label} (Internal Sampler Index Buffer)")) + } else { + Cow::Borrowed("Internal Sampler Index Buffer") + }; + + let buffer_desc = crate::BufferDescriptor { + label: Some(&label), + size: buffer_size, + usage: wgt::BufferUses::STORAGE_READ_ONLY | wgt::BufferUses::MAP_WRITE, + // D3D12 backend doesn't care about the memory flags + memory_flags: crate::MemoryFlags::empty(), + }; + + let (buffer, allocation) = + suballocation::DeviceAllocationContext::from(self).create_buffer(&buffer_desc)?; + + let mut mapping = ptr::null_mut::(); + unsafe { buffer.Map(0, None, Some(&mut mapping)) }.into_device_result("Map")?; + + assert!(!mapping.is_null()); + assert_eq!(mapping as usize % 4, 0); + + unsafe { + ptr::copy_nonoverlapping( + sampler_indexes.as_ptr(), + mapping.cast(), + sampler_indexes.len(), + ) + }; + + // The unmapping is not needed, as all memory is coherent in d3d12, but lets be nice to our address space. + unsafe { buffer.Unmap(0, None) }; + + let srv_desc = Direct3D12::D3D12_SHADER_RESOURCE_VIEW_DESC { + Format: Dxgi::Common::DXGI_FORMAT_UNKNOWN, + ViewDimension: Direct3D12::D3D12_SRV_DIMENSION_BUFFER, + Anonymous: Direct3D12::D3D12_SHADER_RESOURCE_VIEW_DESC_0 { + Buffer: Direct3D12::D3D12_BUFFER_SRV { + FirstElement: 0, + NumElements: sampler_indexes.len() as u32, + StructureByteStride: 4, + Flags: Direct3D12::D3D12_BUFFER_SRV_FLAG_NONE, + }, + }, + Shader4ComponentMapping: Direct3D12::D3D12_DEFAULT_SHADER_4_COMPONENT_MAPPING, + }; + + let inner = cpu_views.as_mut().unwrap(); + let cpu_index = inner.stage.len() as u32; + let srv = desc.layout.cpu_heap_views.as_ref().unwrap().at(cpu_index); + + unsafe { + self.raw + .CreateShaderResourceView(&buffer, Some(&srv_desc), srv) + }; + + cpu_views.as_mut().unwrap().stage.push(srv); + + Some(super::SamplerIndexBuffer { buffer, allocation }) + } else { + None + }; + + let handle_views = match cpu_views { + Some(inner) => { + let dual = unsafe { + descriptor::upload( + &self.raw, + &inner, + &self.shared.heap_views, + &desc.layout.copy_counts, + ) + }?; + Some(dual) + } + None => None, + }; + + self.counters.bind_groups.add(1); + + Ok(super::BindGroup { + handle_views, + sampler_index_buffer, + dynamic_buffers, + }) + } + + unsafe fn destroy_bind_group(&self, group: super::BindGroup) { + if let Some(dual) = group.handle_views { + self.shared.heap_views.free_slice(dual); + } + + if let Some(sampler_buffer) = group.sampler_index_buffer { + suballocation::DeviceAllocationContext::from(self) + .free_resource(sampler_buffer.buffer, sampler_buffer.allocation); + } + + self.counters.bind_groups.sub(1); + } + + unsafe fn create_shader_module( + &self, + desc: &crate::ShaderModuleDescriptor, + shader: crate::ShaderInput, + ) -> Result { + self.counters.shader_modules.add(1); + + let raw_name = desc + .label + .and_then(|label| alloc::ffi::CString::new(label).ok()); + match shader { + crate::ShaderInput::Naga(naga) => Ok(super::ShaderModule { + source: super::ShaderModuleSource::Naga(naga), + raw_name, + runtime_checks: desc.runtime_checks, + }), + crate::ShaderInput::Dxil { + shader, + num_workgroups, + } => Ok(super::ShaderModule { + source: super::ShaderModuleSource::DxilPassthrough(super::DxilPassthroughShader { + shader: shader.to_vec(), + num_workgroups, + }), + raw_name, + runtime_checks: desc.runtime_checks, + }), + crate::ShaderInput::Hlsl { + shader, + num_workgroups, + } => Ok(super::ShaderModule { + source: super::ShaderModuleSource::HlslPassthrough(super::HlslPassthroughShader { + shader: shader.to_owned(), + num_workgroups, + }), + raw_name, + runtime_checks: desc.runtime_checks, + }), + crate::ShaderInput::SpirV(_) + | crate::ShaderInput::MetalLib { .. } + | crate::ShaderInput::Msl { .. } + | crate::ShaderInput::Glsl { .. } => { + unreachable!() + } + } + } + unsafe fn destroy_shader_module(&self, _module: super::ShaderModule) { + self.counters.shader_modules.sub(1); + // just drop + } + + unsafe fn create_render_pipeline( + &self, + desc: &crate::RenderPipelineDescriptor< + super::PipelineLayout, + super::ShaderModule, + super::PipelineCache, + >, + ) -> Result { + let mut shader_stages = wgt::ShaderStages::empty(); + let (topology_class, topology) = conv::map_topology(desc.primitive.topology); + let mut rtv_formats = [Dxgi::Common::DXGI_FORMAT_UNKNOWN; + Direct3D12::D3D12_SIMULTANEOUS_RENDER_TARGET_COUNT as usize]; + for (rtv_format, ct) in rtv_formats.iter_mut().zip(desc.color_targets) { + if let Some(ct) = ct.as_ref() { + *rtv_format = auxil::dxgi::conv::map_texture_format(ct.format); + } + } + + let bias = desc + .depth_stencil + .as_ref() + .map(|ds| ds.bias) + .unwrap_or_default(); + + let rasterizer_state = Direct3D12::D3D12_RASTERIZER_DESC { + FillMode: conv::map_polygon_mode(desc.primitive.polygon_mode), + CullMode: match desc.primitive.cull_mode { + None => Direct3D12::D3D12_CULL_MODE_NONE, + Some(wgt::Face::Front) => Direct3D12::D3D12_CULL_MODE_FRONT, + Some(wgt::Face::Back) => Direct3D12::D3D12_CULL_MODE_BACK, + }, + FrontCounterClockwise: match desc.primitive.front_face { + wgt::FrontFace::Cw => Foundation::FALSE, + wgt::FrontFace::Ccw => Foundation::TRUE, + }, + DepthBias: bias.constant, + DepthBiasClamp: bias.clamp, + SlopeScaledDepthBias: bias.slope_scale, + DepthClipEnable: windows_core::BOOL::from(!desc.primitive.unclipped_depth), + MultisampleEnable: windows_core::BOOL::from(desc.multisample.count > 1), + ForcedSampleCount: 0, + AntialiasedLineEnable: false.into(), + ConservativeRaster: if desc.primitive.conservative { + Direct3D12::D3D12_CONSERVATIVE_RASTERIZATION_MODE_ON + } else { + Direct3D12::D3D12_CONSERVATIVE_RASTERIZATION_MODE_OFF + }, + }; + + let blob_fs = match desc.fragment_stage { + Some(ref stage) => { + shader_stages |= wgt::ShaderStages::FRAGMENT; + Some(self.load_shader(stage, desc.layout, naga::ShaderStage::Fragment, None)?) + } + None => None, + }; + let pixel_shader = match &blob_fs { + Some(shader) => shader.create_native_shader(), + None => Direct3D12::D3D12_SHADER_BYTECODE::default(), + }; + let stream_output = Direct3D12::D3D12_STREAM_OUTPUT_DESC { + pSODeclaration: ptr::null(), + NumEntries: 0, + pBufferStrides: ptr::null(), + NumStrides: 0, + RasterizedStream: 0, + }; + let blend_state = Direct3D12::D3D12_BLEND_DESC { + AlphaToCoverageEnable: windows_core::BOOL::from( + desc.multisample.alpha_to_coverage_enabled, + ), + IndependentBlendEnable: true.into(), + RenderTarget: conv::map_render_targets(desc.color_targets), + }; + let depth_stencil_state = match desc.depth_stencil { + Some(ref ds) => conv::map_depth_stencil(ds), + None => Default::default(), + }; + let dsv_format = desc + .depth_stencil + .as_ref() + .map_or(Dxgi::Common::DXGI_FORMAT_UNKNOWN, |ds| { + auxil::dxgi::conv::map_texture_format(ds.format) + }); + let sample_desc = Dxgi::Common::DXGI_SAMPLE_DESC { + Count: desc.multisample.count, + Quality: 0, + }; + let cached_pso = Direct3D12::D3D12_CACHED_PIPELINE_STATE { + pCachedBlob: ptr::null(), + CachedBlobSizeInBytes: 0, + }; + let flags = Direct3D12::D3D12_PIPELINE_STATE_FLAG_NONE; + + let mut view_instancing = ArrayVec::::new(); + if let Some(mask) = desc.multiview_mask { + let mask = mask.get(); + // This array is just what _could_ be rendered to. We actually apply the mask at + // renderpass creation time. The `view_index` passed to the shader depends on the + // view's index in this array, so if we include every view in this array, `view_index` + // actually the texture array layer, like in vulkan. + for i in 0..32 - mask.leading_zeros() { + view_instancing.push(Direct3D12::D3D12_VIEW_INSTANCE_LOCATION { + ViewportArrayIndex: 0, + RenderTargetArrayIndex: i, + }); + } + } + + // Borrow view instancing slice, so we can be sure that it won't be moved while we have pointers into this buffer. + let view_instancing_slice = view_instancing.as_slice(); + + let mut stream_desc = RenderPipelineStateStreamDesc { + // Shared by vertex and mesh pipelines + root_signature: desc.layout.shared.signature.as_ref(), + pixel_shader, + blend_state, + sample_mask: desc.multisample.mask as u32, + rasterizer_state, + depth_stencil_state, + primitive_topology_type: topology_class, + rtv_formats: Direct3D12::D3D12_RT_FORMAT_ARRAY { + RTFormats: rtv_formats, + NumRenderTargets: desc.color_targets.len() as u32, + }, + dsv_format, + sample_desc, + node_mask: 0, + cached_pso, + flags, + view_instancing: if !view_instancing_slice.is_empty() { + Some(Direct3D12::D3D12_VIEW_INSTANCING_DESC { + ViewInstanceCount: view_instancing_slice.len() as u32, + pViewInstanceLocations: view_instancing_slice.as_ptr(), + // This lets us hide/mask certain values later, at renderpass creation time. + Flags: Direct3D12::D3D12_VIEW_INSTANCING_FLAG_ENABLE_VIEW_INSTANCE_MASKING, + }) + } else { + None + }, + + // Optional data that depends on the pipeline type (vertex vs mesh). + vertex_shader: Default::default(), + input_layout: Default::default(), + index_buffer_strip_cut_value: Default::default(), + stream_output, + task_shader: Default::default(), + mesh_shader: Default::default(), + }; + let mut input_element_descs = Vec::new(); + let blob_vs; + let blob_ts; + let blob_ms; + let mut vertex_strides = [None; crate::MAX_VERTEX_BUFFERS]; + match &desc.vertex_processor { + &crate::VertexProcessor::Standard { + vertex_buffers, + ref vertex_stage, + } => { + shader_stages |= wgt::ShaderStages::VERTEX; + blob_vs = Some(self.load_shader( + vertex_stage, + desc.layout, + naga::ShaderStage::Vertex, + desc.fragment_stage.as_ref(), + )?); + + for (i, (stride, vbuf)) in vertex_strides.iter_mut().zip(vertex_buffers).enumerate() + { + *stride = Some(vbuf.array_stride as u32); + let (slot_class, step_rate) = match vbuf.step_mode { + wgt::VertexStepMode::Vertex => { + (Direct3D12::D3D12_INPUT_CLASSIFICATION_PER_VERTEX_DATA, 0) + } + wgt::VertexStepMode::Instance => { + (Direct3D12::D3D12_INPUT_CLASSIFICATION_PER_INSTANCE_DATA, 1) + } + }; + for attribute in vbuf.attributes { + input_element_descs.push(Direct3D12::D3D12_INPUT_ELEMENT_DESC { + SemanticName: windows::core::PCSTR(NAGA_LOCATION_SEMANTIC.as_ptr()), + SemanticIndex: attribute.shader_location, + Format: auxil::dxgi::conv::map_vertex_format(attribute.format), + InputSlot: i as u32, + AlignedByteOffset: attribute.offset as u32, + InputSlotClass: slot_class, + InstanceDataStepRate: step_rate, + }); + } + } + stream_desc.vertex_shader = blob_vs.as_ref().unwrap().create_native_shader(); + stream_desc.input_layout = Direct3D12::D3D12_INPUT_LAYOUT_DESC { + pInputElementDescs: if input_element_descs.is_empty() { + ptr::null() + } else { + input_element_descs.as_ptr() + }, + NumElements: input_element_descs.len() as u32, + }; + stream_desc.index_buffer_strip_cut_value = match desc.primitive.strip_index_format { + Some(wgt::IndexFormat::Uint16) => { + Direct3D12::D3D12_INDEX_BUFFER_STRIP_CUT_VALUE_0xFFFF + } + Some(wgt::IndexFormat::Uint32) => { + Direct3D12::D3D12_INDEX_BUFFER_STRIP_CUT_VALUE_0xFFFFFFFF + } + None => Direct3D12::D3D12_INDEX_BUFFER_STRIP_CUT_VALUE_DISABLED, + }; + stream_desc.stream_output = Direct3D12::D3D12_STREAM_OUTPUT_DESC { + pSODeclaration: ptr::null(), + NumEntries: 0, + pBufferStrides: ptr::null(), + NumStrides: 0, + RasterizedStream: 0, + }; + } + crate::VertexProcessor::Mesh { + task_stage, + mesh_stage, + } => { + blob_ts = if let Some(ts) = task_stage { + shader_stages |= wgt::ShaderStages::TASK; + Some(self.load_shader( + ts, + desc.layout, + naga::ShaderStage::Task, + desc.fragment_stage.as_ref(), + )?) + } else { + None + }; + let task_shader = if let Some(ts) = &blob_ts { + ts.create_native_shader() + } else { + Default::default() + }; + shader_stages |= wgt::ShaderStages::MESH; + blob_ms = Some(self.load_shader( + mesh_stage, + desc.layout, + naga::ShaderStage::Mesh, + desc.fragment_stage.as_ref(), + )?); + stream_desc.task_shader = task_shader; + stream_desc.mesh_shader = blob_ms.as_ref().unwrap().create_native_shader(); + } + }; + let raw: Direct3D12::ID3D12PipelineState = + // If stream descriptors are available, use them as they are more flexible. + if let Ok(device) = self.raw.cast::() { + // Prefer stream descs where possible + let mut stream = stream_desc.to_stream(); + unsafe { + profiling::scope!("ID3D12Device2::CreatePipelineState"); + stream.create_pipeline_state(&device).map_err(|err| { + crate::PipelineError::Linkage(shader_stages, err.to_string()) + })? + } + } else { + unsafe { + // Safety: `stream_desc` entirely outlives the `desc`. + let desc = stream_desc.to_graphics_pipeline_descriptor(); + self.raw.CreateGraphicsPipelineState(&desc).map_err(|err| { + crate::PipelineError::Linkage(shader_stages, err.to_string()) + })? + } + }; + + if let Some(label) = desc.label { + raw.set_name(label)?; + } + + self.counters.render_pipelines.add(1); + + Ok(super::RenderPipeline { + raw, + layout: desc.layout.shared.clone(), + topology, + vertex_strides, + }) + } + + unsafe fn destroy_render_pipeline(&self, _pipeline: super::RenderPipeline) { + self.counters.render_pipelines.sub(1); + } + + unsafe fn create_compute_pipeline( + &self, + desc: &crate::ComputePipelineDescriptor< + super::PipelineLayout, + super::ShaderModule, + super::PipelineCache, + >, + ) -> Result { + let blob_cs = + self.load_shader(&desc.stage, desc.layout, naga::ShaderStage::Compute, None)?; + + let pair = { + profiling::scope!("ID3D12Device::CreateComputePipelineState"); + unsafe { + self.raw.CreateComputePipelineState( + &Direct3D12::D3D12_COMPUTE_PIPELINE_STATE_DESC { + pRootSignature: borrow_optional_interface_temporarily( + &desc.layout.shared.signature, + ), + CS: blob_cs.create_native_shader(), + NodeMask: 0, + CachedPSO: Direct3D12::D3D12_CACHED_PIPELINE_STATE::default(), + Flags: Direct3D12::D3D12_PIPELINE_STATE_FLAG_NONE, + }, + ) + } + }; + + let raw: Direct3D12::ID3D12PipelineState = pair.map_err(|err| { + crate::PipelineError::Linkage(wgt::ShaderStages::COMPUTE, err.to_string()) + })?; + + if let Some(label) = desc.label { + raw.set_name(label)?; + } + + self.counters.compute_pipelines.add(1); + + Ok(super::ComputePipeline { + raw, + layout: desc.layout.shared.clone(), + }) + } + + unsafe fn destroy_compute_pipeline(&self, _pipeline: super::ComputePipeline) { + self.counters.compute_pipelines.sub(1); + } + + unsafe fn create_pipeline_cache( + &self, + _desc: &crate::PipelineCacheDescriptor<'_>, + ) -> Result { + Ok(super::PipelineCache) + } + unsafe fn destroy_pipeline_cache(&self, _: super::PipelineCache) {} + + unsafe fn create_query_set( + &self, + desc: &wgt::QuerySetDescriptor, + ) -> Result { + let (heap_ty, raw_ty) = match desc.ty { + wgt::QueryType::Occlusion => ( + Direct3D12::D3D12_QUERY_HEAP_TYPE_OCCLUSION, + Direct3D12::D3D12_QUERY_TYPE_BINARY_OCCLUSION, + ), + wgt::QueryType::PipelineStatistics(_) => ( + Direct3D12::D3D12_QUERY_HEAP_TYPE_PIPELINE_STATISTICS, + Direct3D12::D3D12_QUERY_TYPE_PIPELINE_STATISTICS, + ), + wgt::QueryType::Timestamp => ( + Direct3D12::D3D12_QUERY_HEAP_TYPE_TIMESTAMP, + Direct3D12::D3D12_QUERY_TYPE_TIMESTAMP, + ), + }; + + if let Some(threshold) = self + .mem_allocator + .memory_budget_thresholds + .for_resource_creation + { + let info = self + .shared + .adapter + .query_video_memory_info(Dxgi::DXGI_MEMORY_SEGMENT_GROUP_LOCAL)?; + + // Assume each query is 256 bytes. + // On an AMD W6800 with driver version 32.0.12030.9, occlusion and pipeline statistics are 256, timestamp is 8. + + if info.CurrentUsage + desc.count as u64 * 256 >= info.Budget / 100 * threshold as u64 { + return Err(crate::DeviceError::OutOfMemory); + } + } + + let mut raw = None::; + unsafe { + self.raw.CreateQueryHeap( + &Direct3D12::D3D12_QUERY_HEAP_DESC { + Type: heap_ty, + Count: desc.count, + NodeMask: 0, + }, + &mut raw, + ) + } + .into_device_result("Query heap creation")?; + + let raw = raw.ok_or(crate::DeviceError::Unexpected)?; + + if let Some(label) = desc.label { + raw.set_name(label)?; + } + + self.counters.query_sets.add(1); + + Ok(super::QuerySet { raw, raw_ty }) + } + + unsafe fn destroy_query_set(&self, _set: super::QuerySet) { + self.counters.query_sets.sub(1); + } + + unsafe fn create_fence(&self) -> Result { + let raw: Direct3D12::ID3D12Fence = + unsafe { self.raw.CreateFence(0, Direct3D12::D3D12_FENCE_FLAG_SHARED) } + .into_device_result("Fence creation")?; + + self.counters.fences.add(1); + + Ok(super::Fence { raw }) + } + unsafe fn destroy_fence(&self, _fence: super::Fence) { + self.counters.fences.sub(1); + } + + unsafe fn get_fence_value( + &self, + fence: &super::Fence, + ) -> Result { + Ok(unsafe { fence.raw.GetCompletedValue() }) + } + unsafe fn wait( + &self, + fence: &super::Fence, + value: crate::FenceValue, + timeout: Option, + ) -> Result { + let timeout = timeout.unwrap_or(Duration::MAX); + + // We first check if the fence has already reached the value we're waiting for. + let mut fence_value = unsafe { fence.raw.GetCompletedValue() }; + if fence_value >= value { + return Ok(true); + } + + let event = Event::create(false, false)?; + + unsafe { fence.raw.SetEventOnCompletion(value, event.0) } + .into_device_result("Set event")?; + + let start_time = Instant::now(); + + // We need to loop to get correct behavior when timeouts are involved. + // + // wait(0): + // - We set the event from the fence value 0. + // - WaitForSingleObject times out, we return false. + // + // wait(1): + // - We set the event from the fence value 1. + // - WaitForSingleObject returns. However we do not know if the fence value is 0 or 1, + // just that _something_ triggered the event. We check the fence value, and if it is + // 1, we return true. Otherwise, we loop and wait again. + loop { + let elapsed = start_time.elapsed(); + + // We need to explicitly use checked_sub. Overflow with duration panics, and if the + // timing works out just right, we can get a negative remaining wait duration. + // + // This happens when a previous iteration WaitForSingleObject succeeded with a previous fence value, + // right before the timeout would have been hit. + let remaining_wait_duration = match timeout.checked_sub(elapsed) { + Some(remaining) => remaining, + None => { + log::trace!("Timeout elapsed in between waits!"); + break Ok(false); + } + }; + + log::trace!("Waiting for fence value {value} for {remaining_wait_duration:?}"); + + match unsafe { + Threading::WaitForSingleObject( + event.0, + remaining_wait_duration.as_millis().min(u32::MAX as u128) as u32, + ) + } { + Foundation::WAIT_OBJECT_0 => {} + Foundation::WAIT_ABANDONED | Foundation::WAIT_FAILED => { + log::error!("Wait failed!"); + break Err(crate::DeviceError::Lost); + } + Foundation::WAIT_TIMEOUT => { + log::trace!("Wait timed out!"); + break Ok(false); + } + other => { + log::error!("Unexpected wait status: 0x{other:?}"); + break Err(crate::DeviceError::Lost); + } + }; + + fence_value = unsafe { fence.raw.GetCompletedValue() }; + log::trace!("Wait complete! Fence actual value: {fence_value}"); + + if fence_value >= value { + break Ok(true); + } + } + } + + unsafe fn start_graphics_debugger_capture(&self) -> bool { + #[cfg(feature = "renderdoc")] + { + unsafe { + self.render_doc + .start_frame_capture(self.raw.as_raw(), ptr::null_mut()) + } + } + #[cfg(not(feature = "renderdoc"))] + false + } + + unsafe fn stop_graphics_debugger_capture(&self) { + #[cfg(feature = "renderdoc")] + unsafe { + self.render_doc + .end_frame_capture(self.raw.as_raw(), ptr::null_mut()) + } + } + + unsafe fn get_acceleration_structure_build_sizes<'a>( + &self, + desc: &crate::GetAccelerationStructureBuildSizesDescriptor<'a, super::Buffer>, + ) -> crate::AccelerationStructureBuildSizes { + let mut geometry_desc; + let device5 = self.raw.cast::().unwrap(); + let ty; + let inputs0; + let num_desc; + match desc.entries { + AccelerationStructureEntries::Instances(instances) => { + ty = Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_TYPE_TOP_LEVEL; + inputs0 = Direct3D12::D3D12_BUILD_RAYTRACING_ACCELERATION_STRUCTURE_INPUTS_0 { + InstanceDescs: 0, + }; + num_desc = instances.count; + } + AccelerationStructureEntries::Triangles(triangles) => { + geometry_desc = Vec::with_capacity(triangles.len()); + for triangle in triangles { + let index_format = triangle + .indices + .as_ref() + .map_or(Dxgi::Common::DXGI_FORMAT_UNKNOWN, |indices| { + auxil::dxgi::conv::map_index_format(indices.format) + }); + let index_count = triangle.indices.as_ref().map_or(0, |indices| indices.count); + + let triangle_desc = Direct3D12::D3D12_RAYTRACING_GEOMETRY_TRIANGLES_DESC { + // https://learn.microsoft.com/en-us/windows/win32/api/d3d12/nf-d3d12-id3d12device5-getraytracingaccelerationstructureprebuildinfo + // It may not inspect/dereference any GPU virtual addresses, other than + // to check to see if a pointer is NULL or not, such as the optional + // transform in D3D12_RAYTRACING_GEOMETRY_TRIANGLES_DESC, without + // dereferencing it. + // + // This suggests we could pass a non-zero invalid address here if fetching the + // real address has significant overhead, but we pass the real one to be on the + // safe side for now. + Transform3x4: if desc + .flags + .contains(wgt::AccelerationStructureFlags::USE_TRANSFORM) + { + unsafe { + triangle + .transform + .as_ref() + .unwrap() + .buffer + .resource + .GetGPUVirtualAddress() + } + } else { + 0 + }, + IndexFormat: index_format, + VertexFormat: auxil::dxgi::conv::map_vertex_format(triangle.vertex_format), + IndexCount: index_count, + VertexCount: triangle.vertex_count, + IndexBuffer: 0, + VertexBuffer: Direct3D12::D3D12_GPU_VIRTUAL_ADDRESS_AND_STRIDE { + StartAddress: 0, + StrideInBytes: triangle.vertex_stride, + }, + }; + + geometry_desc.push(Direct3D12::D3D12_RAYTRACING_GEOMETRY_DESC { + Type: Direct3D12::D3D12_RAYTRACING_GEOMETRY_TYPE_TRIANGLES, + Flags: conv::map_acceleration_structure_geometry_flags(triangle.flags), + Anonymous: Direct3D12::D3D12_RAYTRACING_GEOMETRY_DESC_0 { + Triangles: triangle_desc, + }, + }) + } + ty = Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_TYPE_BOTTOM_LEVEL; + inputs0 = Direct3D12::D3D12_BUILD_RAYTRACING_ACCELERATION_STRUCTURE_INPUTS_0 { + pGeometryDescs: geometry_desc.as_ptr(), + }; + num_desc = geometry_desc.len() as u32; + } + AccelerationStructureEntries::AABBs(aabbs) => { + geometry_desc = Vec::with_capacity(aabbs.len()); + for aabb in aabbs { + let aabb_desc = Direct3D12::D3D12_RAYTRACING_GEOMETRY_AABBS_DESC { + AABBCount: aabb.count as u64, + AABBs: Direct3D12::D3D12_GPU_VIRTUAL_ADDRESS_AND_STRIDE { + StartAddress: 0, + StrideInBytes: aabb.stride, + }, + }; + geometry_desc.push(Direct3D12::D3D12_RAYTRACING_GEOMETRY_DESC { + Type: Direct3D12::D3D12_RAYTRACING_GEOMETRY_TYPE_PROCEDURAL_PRIMITIVE_AABBS, + Flags: conv::map_acceleration_structure_geometry_flags(aabb.flags), + Anonymous: Direct3D12::D3D12_RAYTRACING_GEOMETRY_DESC_0 { + AABBs: aabb_desc, + }, + }) + } + ty = Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_TYPE_BOTTOM_LEVEL; + inputs0 = Direct3D12::D3D12_BUILD_RAYTRACING_ACCELERATION_STRUCTURE_INPUTS_0 { + pGeometryDescs: geometry_desc.as_ptr(), + }; + num_desc = geometry_desc.len() as u32; + } + }; + let acceleration_structure_inputs = + Direct3D12::D3D12_BUILD_RAYTRACING_ACCELERATION_STRUCTURE_INPUTS { + Type: ty, + Flags: conv::map_acceleration_structure_build_flags(desc.flags, None), + NumDescs: num_desc, + DescsLayout: Direct3D12::D3D12_ELEMENTS_LAYOUT_ARRAY, + Anonymous: inputs0, + }; + let mut info = Direct3D12::D3D12_RAYTRACING_ACCELERATION_STRUCTURE_PREBUILD_INFO::default(); + unsafe { + device5.GetRaytracingAccelerationStructurePrebuildInfo( + &acceleration_structure_inputs, + &mut info, + ) + }; + crate::AccelerationStructureBuildSizes { + acceleration_structure_size: info.ResultDataMaxSizeInBytes, + update_scratch_size: info.UpdateScratchDataSizeInBytes, + build_scratch_size: info.ScratchDataSizeInBytes, + } + } + + unsafe fn get_acceleration_structure_device_address( + &self, + acceleration_structure: &super::AccelerationStructure, + ) -> wgt::BufferAddress { + unsafe { acceleration_structure.resource.GetGPUVirtualAddress() } + } + + unsafe fn create_acceleration_structure( + &self, + desc: &crate::AccelerationStructureDescriptor, + ) -> Result { + // Create a D3D12 resource as per-usual. + let size = desc.size; + + let raw_desc = Direct3D12::D3D12_RESOURCE_DESC { + Dimension: Direct3D12::D3D12_RESOURCE_DIMENSION_BUFFER, + Alignment: 0, + Width: size, + Height: 1, + DepthOrArraySize: 1, + MipLevels: 1, + Format: Dxgi::Common::DXGI_FORMAT_UNKNOWN, + SampleDesc: Dxgi::Common::DXGI_SAMPLE_DESC { + Count: 1, + Quality: 0, + }, + Layout: Direct3D12::D3D12_TEXTURE_LAYOUT_ROW_MAJOR, + // TODO: when moving to enhanced barriers use Direct3D12::D3D12_RESOURCE_FLAG_RAYTRACING_ACCELERATION_STRUCTURE + Flags: Direct3D12::D3D12_RESOURCE_FLAG_ALLOW_UNORDERED_ACCESS, + }; + + let (resource, allocation) = suballocation::DeviceAllocationContext::from(self) + .create_acceleration_structure(desc, raw_desc)?; + + // for some reason there is no counter for acceleration structures + + Ok(super::AccelerationStructure { + resource, + allocation, + }) + } + + unsafe fn destroy_acceleration_structure( + &self, + acceleration_structure: super::AccelerationStructure, + ) { + suballocation::DeviceAllocationContext::from(self).free_resource( + acceleration_structure.resource, + acceleration_structure.allocation, + ); + } + + fn get_internal_counters(&self) -> wgt::HalCounters { + self.counters.as_ref().clone() + } + + fn generate_allocator_report(&self) -> Option { + Some(self.mem_allocator.generate_report()) + } + + fn tlas_instance_to_bytes(&self, instance: TlasInstance) -> Vec { + const MAX_U24: u32 = (1u32 << 24u32) - 1u32; + let temp = Direct3D12::D3D12_RAYTRACING_INSTANCE_DESC { + Transform: instance.transform, + _bitfield1: (instance.custom_data & MAX_U24) | (u32::from(instance.mask) << 24), + _bitfield2: 0, + AccelerationStructure: instance.blas_address, + }; + + wgt::bytemuck_wrapper!(unsafe struct Desc(Direct3D12::D3D12_RAYTRACING_INSTANCE_DESC)); + + bytemuck::bytes_of(&Desc::wrap(temp)).to_vec() + } + + fn check_if_oom(&self) -> Result<(), crate::DeviceError> { + let Some(threshold) = self.mem_allocator.memory_budget_thresholds.for_device_loss else { + return Ok(()); + }; + + let info = self + .shared + .adapter + .query_video_memory_info(Dxgi::DXGI_MEMORY_SEGMENT_GROUP_LOCAL)?; + + if info.CurrentUsage >= info.Budget / 100 * threshold as u64 { + return Err(crate::DeviceError::OutOfMemory); + } + + if matches!( + self.shared.private_caps.memory_architecture, + super::MemoryArchitecture::NonUnified + ) { + let info = self + .shared + .adapter + .query_video_memory_info(Dxgi::DXGI_MEMORY_SEGMENT_GROUP_NON_LOCAL)?; + + if info.CurrentUsage >= info.Budget / 100 * threshold as u64 { + return Err(crate::DeviceError::OutOfMemory); + } + } + + Ok(()) + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/device_creation.rs b/third_party/wgpu-hal-29.0.4/src/dx12/device_creation.rs new file mode 100644 index 0000000..dc5cdf9 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/device_creation.rs @@ -0,0 +1,176 @@ +use alloc::sync::Arc; +use core::ops::Deref; + +use windows::core::Interface as _; +use windows::Win32::Graphics::{Direct3D, Direct3D12}; + +use super::D3D12Lib; +use crate::auxil::dxgi::factory::DxgiAdapter; + +/// Abstraction over D3D12 device creation. +/// +/// Supports two paths: +/// - **Independent**: Uses `ID3D12DeviceFactory` from the Agility SDK's Independent Devices API. +/// - **Legacy**: Uses the traditional `D3D12CreateDevice` export. +pub(super) enum DeviceFactory { + /// Uses `ID3D12DeviceFactory` from the Independent Devices API. + Independent(Direct3D12::ID3D12DeviceFactory), + /// Uses the traditional `D3D12CreateDevice` export. + Legacy, +} + +impl DeviceFactory { + /// Create a new `DeviceFactory`. + /// + /// If `agility_sdk` is `Some`, attempts to set up the Independent Devices API path. + /// On failure, the behavior depends on + /// [`on_load_failure`](wgt::Dx12AgilitySDKLoadFailure): + /// - [`Fallback`](wgt::Dx12AgilitySDKLoadFailure::Fallback): logs a warning and + /// returns `Ok(Legacy)`. + /// - [`Error`](wgt::Dx12AgilitySDKLoadFailure::Error): returns an `Err`. + pub(super) fn new( + lib: &D3D12Lib, + agility_sdk: Option<&wgt::Dx12AgilitySDK>, + ) -> Result { + let Some(agility_sdk) = agility_sdk else { + log::debug!("No D3D12 Agility SDK configuration provided; using system D3D12 runtime"); + return Ok(Self::Legacy); + }; + + match Self::try_create_independent(lib, agility_sdk) { + Ok(factory) => { + log::debug!( + "Using D3D12 Agility SDK v{} from '{}'", + agility_sdk.sdk_version, + agility_sdk.sdk_path + ); + Ok(Self::Independent(factory)) + } + Err(err) => { + let message = format!( + "Failed to initialize D3D12 Agility SDK (v{} at '{}'): {err}", + agility_sdk.sdk_version, agility_sdk.sdk_path + ); + + match agility_sdk.on_load_failure { + wgt::Dx12AgilitySDKLoadFailure::Fallback => { + log::warn!("{message}; falling back to system D3D12 runtime"); + Ok(Self::Legacy) + } + wgt::Dx12AgilitySDKLoadFailure::Error => { + Err(crate::InstanceError::new(message)) + } + } + } + } + } + + fn try_create_independent( + lib: &D3D12Lib, + agility_sdk: &wgt::Dx12AgilitySDK, + ) -> Result { + // Step 1: Get ID3D12SDKConfiguration1 via D3D12GetInterface + let sdk_config: Direct3D12::ID3D12SDKConfiguration1 = lib + .get_interface(&Direct3D12::CLSID_D3D12SDKConfiguration) + .map_err(DeviceFactoryError::GetInterface)?; + + // Step 2: Create device factory with the specified SDK version and path + let sdk_path = std::ffi::CString::new(agility_sdk.sdk_path.as_bytes()) + .map_err(|_| DeviceFactoryError::InvalidPath)?; + let factory: Direct3D12::ID3D12DeviceFactory = unsafe { + sdk_config.CreateDeviceFactory( + agility_sdk.sdk_version, + windows::core::PCSTR(sdk_path.as_ptr().cast::()), + ) + } + .map_err(DeviceFactoryError::CreateDeviceFactory)?; + + Ok(factory) + } + + /// Enable the D3D12 debug layer and optionally GPU-based validation. + /// + /// - **Legacy**: configures debug globally via `D3D12GetDebugInterface`. + /// - **Independent**: uses `GetConfigurationInterface` to get an + /// `ID3D12Debug` scoped to the factory. + pub(super) fn enable_debug_layer(&self, lib: &D3D12Lib, flags: wgt::InstanceFlags) { + if !flags + .intersects(wgt::InstanceFlags::VALIDATION | wgt::InstanceFlags::GPU_BASED_VALIDATION) + { + return; + } + + let debug_controller = match self { + Self::Independent(factory) => { + match unsafe { + factory.GetConfigurationInterface::( + &Direct3D12::CLSID_D3D12Debug, + ) + } { + Ok(debug) => debug, + Err(err) => { + log::warn!("Failed to get debug interface from device factory: {err}"); + return; + } + } + } + Self::Legacy => match lib.debug_interface() { + Ok(Some(debug)) => debug, + Ok(None) => return, + Err(err) => { + log::warn!("Failed to get debug interface: {err}"); + return; + } + }, + }; + + if flags.intersects(wgt::InstanceFlags::VALIDATION) { + unsafe { debug_controller.EnableDebugLayer() } + } + if flags.intersects(wgt::InstanceFlags::GPU_BASED_VALIDATION) { + if let Ok(debug1) = debug_controller.cast::() { + unsafe { debug1.SetEnableGPUBasedValidation(true) } + } else { + log::warn!("Failed to enable GPU-based validation"); + } + } + } + + /// Create a D3D12 device using the appropriate method. + pub(super) fn create_device( + &self, + lib: &Arc, + adapter: &DxgiAdapter, + feature_level: Direct3D::D3D_FEATURE_LEVEL, + ) -> Result { + match self { + Self::Independent(factory) => { + let mut result__: Option = None; + unsafe { factory.CreateDevice(adapter.deref(), feature_level, &mut result__) } + .map_err(|e| super::CreateDeviceError::D3D12CreateDevice(e.into()))?; + + result__.ok_or(super::CreateDeviceError::RetDeviceIsNull) + } + Self::Legacy => lib.create_device(adapter, feature_level), + } + } +} + +impl core::fmt::Debug for DeviceFactory { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::Independent(_) => write!(f, "DeviceFactory::Independent"), + Self::Legacy => write!(f, "DeviceFactory::Legacy"), + } + } +} + +#[derive(Debug, thiserror::Error)] +enum DeviceFactoryError { + #[error("failed to get ID3D12SDKConfiguration1: {0}")] + GetInterface(super::GetInterfaceError), + #[error("SDK path contains null bytes")] + InvalidPath, + #[error("CreateDeviceFactory failed: {0}")] + CreateDeviceFactory(windows::core::Error), +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/instance.rs b/third_party/wgpu-hal-29.0.4/src/dx12/instance.rs new file mode 100644 index 0000000..6c3a33b --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/instance.rs @@ -0,0 +1,188 @@ +use alloc::{string::String, sync::Arc, vec::Vec}; + +use parking_lot::RwLock; +use windows::Win32::{Foundation, Graphics::Dxgi}; + +use super::SurfaceTarget; +use crate::{ + auxil, + dx12::{ + device_creation::DeviceFactory, shader_compilation::CompilerContainer, D3D12Lib, DCompLib, + }, +}; + +impl crate::Instance for super::Instance { + type A = super::Api; + + unsafe fn init(desc: &crate::InstanceDescriptor<'_>) -> Result { + profiling::scope!("Init DX12 Backend"); + let lib_main = D3D12Lib::new().map_err(|e| { + crate::InstanceError::with_source(String::from("failed to load d3d12.dll"), e) + })?; + + // Create DeviceFactory first so we know which debug path to use + let device_factory = + DeviceFactory::new(&lib_main, desc.backend_options.dx12.agility_sdk.as_ref())?; + + device_factory.enable_debug_layer(&lib_main, desc.flags); + + let (lib_dxgi, factory) = auxil::dxgi::factory::create_factory(desc.flags)?; + + // Create IDXGIFactoryMedia + let factory_media = lib_dxgi.create_factory_media().ok(); + + let mut supports_allow_tearing = false; + if let Some(factory5) = factory.as_factory5() { + let mut allow_tearing = Foundation::FALSE; + let hr = unsafe { + factory5.CheckFeatureSupport( + Dxgi::DXGI_FEATURE_PRESENT_ALLOW_TEARING, + <*mut _>::cast(&mut allow_tearing), + size_of_val(&allow_tearing) as u32, + ) + }; + + match hr { + Err(err) => log::warn!("Unable to check for tearing support: {err}"), + Ok(()) => supports_allow_tearing = true, + } + } + + // Initialize the shader compiler + let compiler_container = match desc.backend_options.dx12.shader_compiler.clone() { + wgt::Dx12Compiler::DynamicDxc { dxc_path } => { + CompilerContainer::new_dynamic_dxc(dxc_path.into()).map_err(|e| { + crate::InstanceError::with_source(String::from("Failed to load dynamic DXC"), e) + })? + } + wgt::Dx12Compiler::StaticDxc => CompilerContainer::new_static_dxc().map_err(|e| { + crate::InstanceError::with_source(String::from("Failed to load static DXC"), e) + })?, + wgt::Dx12Compiler::Fxc => CompilerContainer::new_fxc().map_err(|e| { + crate::InstanceError::with_source(String::from("Failed to load FXC"), e) + })?, + wgt::Dx12Compiler::Auto => { + if cfg!(feature = "static-dxc") { + // Prefer static DXC if its compiled in + CompilerContainer::new_static_dxc().map_err(|e| { + crate::InstanceError::with_source( + String::from("Failed to load static DXC"), + e, + ) + })? + } else { + // Try to load dynamic DXC + let dynamic = CompilerContainer::new_dynamic_dxc("dxcompiler.dll".into()); + match dynamic { + Ok(v) => v, + Err(super::shader_compilation::GetContainerError::FailedToLoad(..)) => { + // If it can't be found load FXC + CompilerContainer::new_fxc().map_err(|e| { + crate::InstanceError::with_source( + String::from("Failed to load FXC"), + e, + ) + })? + } + Err(e) => { + // If another error occurs when loading static DXC return that error + return Err(crate::InstanceError::with_source( + String::from("Failed to load dynamic DXC"), + e, + )); + } + } + } + } + }; + + match compiler_container { + CompilerContainer::DynamicDxc(..) => { + log::debug!("Using dynamic DXC for shader compilation") + } + CompilerContainer::StaticDxc(..) => { + log::debug!("Using static DXC for shader compilation") + } + CompilerContainer::Fxc(..) => { + log::debug!("Using FXC for shader compilation") + } + } + + Ok(Self { + // The call to create_factory will only succeed if we get a factory4, so this is safe. + factory, + factory_media, + library: Arc::new(lib_main), + device_factory: Arc::new(device_factory), + dcomp_lib: Arc::new(DCompLib::new()), + presentation_system: desc.backend_options.dx12.presentation_system, + _lib_dxgi: lib_dxgi, + supports_allow_tearing, + flags: desc.flags, + memory_budget_thresholds: desc.memory_budget_thresholds, + compiler_container: Arc::new(compiler_container), + options: desc.backend_options.dx12.clone(), + telemetry: desc.telemetry, + }) + } + + unsafe fn create_surface( + &self, + display_handle: raw_window_handle::RawDisplayHandle, + window_handle: raw_window_handle::RawWindowHandle, + ) -> Result { + assert!(matches!( + display_handle, + raw_window_handle::RawDisplayHandle::Windows(_) + )); + match window_handle { + raw_window_handle::RawWindowHandle::Win32(handle) => { + // https://github.com/rust-windowing/raw-window-handle/issues/171 + let handle = Foundation::HWND(handle.hwnd.get() as *mut _); + let target = match self.presentation_system { + wgt::Dx12SwapchainKind::DxgiFromHwnd => SurfaceTarget::WndHandle(handle), + wgt::Dx12SwapchainKind::DxgiFromVisual => SurfaceTarget::VisualFromWndHandle { + handle, + dcomp_state: Default::default(), + }, + }; + + Ok(super::Surface { + factory: self.factory.clone(), + factory_media: self.factory_media.clone(), + target, + supports_allow_tearing: self.supports_allow_tearing, + swap_chain: RwLock::new(None), + options: self.options.clone(), + }) + } + _ => Err(crate::InstanceError::new(format!( + "window handle {window_handle:?} is not a Win32 handle" + ))), + } + } + + unsafe fn enumerate_adapters( + &self, + _surface_hint: Option<&super::Surface>, + ) -> Vec> { + let adapters = auxil::dxgi::factory::enumerate_adapters(self.factory.clone()); + + adapters + .into_iter() + .filter_map(|raw| { + super::Adapter::expose( + raw, + &self.library, + &self.device_factory, + &self.dcomp_lib, + self.flags, + self.memory_budget_thresholds, + self.compiler_container.clone(), + self.options.clone(), + self.telemetry, + ) + }) + .collect() + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/mod.rs b/third_party/wgpu-hal-29.0.4/src/dx12/mod.rs new file mode 100644 index 0000000..ac520b2 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/mod.rs @@ -0,0 +1,1698 @@ +/*! +# DirectX12 API internals. + +Generally the mapping is straightforward. + +## Resource transitions + +D3D12 API matches WebGPU internal states very well. The only +caveat here is issuing a special UAV barrier whenever both source +and destination states match, and they are for storage sync. + +## Memory + +For now, all resources are created with "committed" memory. + +## Sampler Descriptor Management + +At most one descriptor heap of each type can be bound at once. This +means that the descriptors from all bind groups need to be present +in the same heap, and they need to be contiguous within that heap. +This is not a problem for the SRV/CBV/UAV heap as it can be sized into +the millions of entries. However the sampler heap is limited to 2048 entries. + +In order to work around this limitation, we refer to samplers indirectly by index. +The entire sampler heap is bound at once and a buffer containing all sampler indexes +for that bind group is bound. The shader then uses the index to look up the sampler +in the heap. To help visualize this, the generated HLSL looks like this: + +```wgsl +@group(0) @binding(2) var myLinearSampler: sampler; +@group(1) @binding(1) var myAnisoSampler: sampler; +@group(1) @binding(4) var myCompSampler: sampler; +``` + +```cpp +// These bindings alias the same descriptors. Depending on the type, the shader will use the correct one. +SamplerState nagaSamplerHeap[2048]: register(s0, space0); +SamplerComparisonState nagaComparisonSamplerHeap[2048]: register(s2048, space1); + +StructuredBuffer nagaGroup0SamplerIndexArray : register(t0, space0); +StructuredBuffer nagaGroup1SamplerIndexArray : register(t1, space0); + +// Indexes into group 0 index array +static const SamplerState myLinearSampler = nagaSamplerHeap[nagaGroup0SamplerIndexArray[0]]; + +// Indexes into group 1 index array +static const SamplerState myAnisoSampler = nagaSamplerHeap[nagaGroup1SamplerIndexArray[0]]; +static const SamplerComparisonState myCompSampler = nagaComparisonSamplerHeap[nagaGroup1SamplerIndexArray[1]]; +``` + +Without this transform we would need separate set of sampler descriptors for each unique combination of samplers +in a bind group. This results in a lot of duplication and makes it easy to hit the 2048 limit. With the transform +the limit is merely 2048 unique samplers in existence, which is much more reasonable. + +## Resource binding + +See [`crate::Device::create_pipeline_layout`] documentation for the structure +of the root signature corresponding to WebGPU pipeline layout. + +Binding groups is mostly straightforward, with one big caveat: +all bindings have to be reset whenever the root signature changes. +This is the rule of D3D12, and we can do nothing to help it. + +We detect this change at both [`crate::CommandEncoder::set_bind_group`] +and [`crate::CommandEncoder::set_render_pipeline`] with +[`crate::CommandEncoder::set_compute_pipeline`]. + +For this reason, in order avoid repeating the binding code, +we are binding everything in `CommandEncoder::update_root_elements`. +When the pipeline layout is changed, we reset all bindings. +Otherwise, we pass a range corresponding only to the current bind group. + +!*/ + +mod adapter; +mod command; +mod conv; +mod dcomp; +mod descriptor; +mod device; +mod device_creation; +mod instance; +mod pipeline_desc; +mod sampler; +mod shader_compilation; +mod suballocation; +mod types; +mod view; + +use alloc::{borrow::ToOwned as _, string::String, sync::Arc, vec::Vec}; +use core::{ffi, fmt, mem, ops::Deref}; + +use arrayvec::ArrayVec; +use hashbrown::HashMap; +use parking_lot::{Mutex, RwLock}; +use suballocation::Allocator; +use windows::{ + core::{Free as _, Interface}, + Win32::{ + Foundation, + Graphics::{ + Direct3D, + Direct3D12::{self, D3D12_TEXTURE_DATA_PLACEMENT_ALIGNMENT}, + DirectComposition, Dxgi, + }, + System::Threading, + }, +}; + +use self::dcomp::DCompLib; +use crate::auxil::{ + self, + dxgi::{ + factory::{DxgiAdapter, DxgiFactory}, + result::HResult, + }, +}; + +#[derive(Debug)] +struct DynLib { + inner: libloading::Library, +} + +impl DynLib { + unsafe fn new

(filename: P) -> Result + where + P: AsRef, + { + unsafe { libloading::Library::new(filename) }.map(|inner| Self { inner }) + } + + unsafe fn get( + &self, + symbol: &[u8], + ) -> Result, crate::DeviceError> { + unsafe { self.inner.get(symbol) }.map_err(|e| match e { + libloading::Error::GetProcAddress { .. } | libloading::Error::GetProcAddressUnknown => { + crate::DeviceError::Unexpected + } + libloading::Error::IncompatibleSize + | libloading::Error::CreateCString { .. } + | libloading::Error::CreateCStringWithTrailing { .. } => crate::hal_internal_error(e), + _ => crate::DeviceError::Unexpected, // could be unreachable!() but we prefer to be more robust + }) + } +} + +#[derive(Debug)] +struct D3D12Lib { + lib: DynLib, +} + +#[derive(Clone, Copy)] +pub enum CreateDeviceError { + GetProcAddress, + D3D12CreateDevice(windows_core::HRESULT), + RetDeviceIsNull, +} + +impl D3D12Lib { + fn new() -> Result { + unsafe { DynLib::new("d3d12.dll").map(|lib| Self { lib }) } + } + + fn create_device( + &self, + adapter: &DxgiAdapter, + feature_level: Direct3D::D3D_FEATURE_LEVEL, + ) -> Result { + // Calls windows::Win32::Graphics::Direct3D12::D3D12CreateDevice on d3d12.dll + type Fun = extern "system" fn( + padapter: *mut ffi::c_void, + minimumfeaturelevel: Direct3D::D3D_FEATURE_LEVEL, + riid: *const windows_core::GUID, + ppdevice: *mut *mut ffi::c_void, + ) -> windows_core::HRESULT; + let func: libloading::Symbol = + unsafe { self.lib.get(c"D3D12CreateDevice".to_bytes()) } + .map_err(|_| CreateDeviceError::GetProcAddress)?; + + let mut result__: Option = None; + + let res = (func)( + adapter.as_raw(), + feature_level, + // TODO: Generic? + &Direct3D12::ID3D12Device::IID, + <*mut _>::cast(&mut result__), + ); + + if res.is_err() { + return Err(CreateDeviceError::D3D12CreateDevice(res)); + } + + result__.ok_or(CreateDeviceError::RetDeviceIsNull) + } + + fn serialize_root_signature( + &self, + version: Direct3D12::D3D_ROOT_SIGNATURE_VERSION, + parameters: &[Direct3D12::D3D12_ROOT_PARAMETER], + static_samplers: &[Direct3D12::D3D12_STATIC_SAMPLER_DESC], + flags: Direct3D12::D3D12_ROOT_SIGNATURE_FLAGS, + ) -> Result { + // Calls windows::Win32::Graphics::Direct3D12::D3D12SerializeRootSignature on d3d12.dll + type Fun = extern "system" fn( + prootsignature: *const Direct3D12::D3D12_ROOT_SIGNATURE_DESC, + version: Direct3D12::D3D_ROOT_SIGNATURE_VERSION, + ppblob: *mut *mut ffi::c_void, + pperrorblob: *mut *mut ffi::c_void, + ) -> windows_core::HRESULT; + let func: libloading::Symbol = + unsafe { self.lib.get(c"D3D12SerializeRootSignature".to_bytes()) }?; + + let desc = Direct3D12::D3D12_ROOT_SIGNATURE_DESC { + NumParameters: parameters.len() as _, + pParameters: parameters.as_ptr(), + NumStaticSamplers: static_samplers.len() as _, + pStaticSamplers: static_samplers.as_ptr(), + Flags: flags, + }; + + let mut blob = None; + let mut error = None::; + (func)( + &desc, + version, + <*mut _>::cast(&mut blob), + <*mut _>::cast(&mut error), + ) + .ok() + .into_device_result("Root signature serialization")?; + + if let Some(error) = error { + let error = D3DBlob(error); + log::error!( + "Root signature serialization error: {:?}", + unsafe { error.as_c_str() }.unwrap().to_str().unwrap() + ); + return Err(crate::DeviceError::Unexpected); // could be hal_usage_error or hal_internal_error + } + + blob.ok_or(crate::DeviceError::Unexpected) + } + + fn debug_interface(&self) -> Result, crate::DeviceError> { + // Calls windows::Win32::Graphics::Direct3D12::D3D12GetDebugInterface on d3d12.dll + type Fun = extern "system" fn( + riid: *const windows_core::GUID, + ppvdebug: *mut *mut ffi::c_void, + ) -> windows_core::HRESULT; + let func: libloading::Symbol = + unsafe { self.lib.get(c"D3D12GetDebugInterface".to_bytes()) }?; + + let mut result__ = None; + + let res = (func)(&Direct3D12::ID3D12Debug::IID, <*mut _>::cast(&mut result__)).ok(); + + if let Err(ref err) = res { + match err.code() { + Dxgi::DXGI_ERROR_SDK_COMPONENT_MISSING => return Ok(None), + _ => {} + } + } + + res.into_device_result("GetDebugInterface")?; + + result__.ok_or(crate::DeviceError::Unexpected).map(Some) + } + + /// Calls D3D12GetInterface to obtain a COM interface by CLSID and IID. + /// + /// This is used by the Independent Devices API to obtain `ID3D12SDKConfiguration1`. + fn get_interface( + &self, + clsid: &windows_core::GUID, + ) -> Result { + // Calls windows::Win32::Graphics::Direct3D12::D3D12GetInterface on d3d12.dll + type Fun = extern "system" fn( + rclsid: *const windows_core::GUID, + riid: *const windows_core::GUID, + ppvdebug: *mut *mut ffi::c_void, + ) -> windows_core::HRESULT; + let func: libloading::Symbol = + unsafe { self.lib.get(c"D3D12GetInterface".to_bytes()) } + .map_err(|_| GetInterfaceError::GetProcAddress)?; + + let mut result__: Option = None; + + let res = (func)(clsid, &T::IID, <*mut _>::cast(&mut result__)); + + if res.is_err() { + return Err(GetInterfaceError::D3D12GetInterface(res)); + } + + result__.ok_or(GetInterfaceError::RetIsNull) + } +} + +#[derive(Clone, Copy, Debug)] +pub(super) enum GetInterfaceError { + GetProcAddress, + D3D12GetInterface(windows_core::HRESULT), + RetIsNull, +} + +impl fmt::Display for GetInterfaceError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::GetProcAddress => write!(f, "D3D12GetInterface not found in d3d12.dll"), + Self::D3D12GetInterface(hr) => write!(f, "D3D12GetInterface failed: {hr}"), + Self::RetIsNull => write!(f, "D3D12GetInterface returned null"), + } + } +} + +impl core::error::Error for GetInterfaceError {} + +#[derive(Debug)] +pub(super) struct DxgiLib { + lib: DynLib, +} + +impl DxgiLib { + pub fn new() -> Result { + unsafe { DynLib::new("dxgi.dll").map(|lib| Self { lib }) } + } + + /// Will error with crate::DeviceError::Unexpected if DXGI 1.3 is not available. + pub fn debug_interface1(&self) -> Result, crate::DeviceError> { + // Calls windows::Win32::Graphics::Dxgi::DXGIGetDebugInterface1 on dxgi.dll + type Fun = extern "system" fn( + flags: u32, + riid: *const windows_core::GUID, + pdebug: *mut *mut ffi::c_void, + ) -> windows_core::HRESULT; + let func: libloading::Symbol = + unsafe { self.lib.get(c"DXGIGetDebugInterface1".to_bytes()) }?; + + let mut result__ = None; + + let res = (func)(0, &Dxgi::IDXGIInfoQueue::IID, <*mut _>::cast(&mut result__)).ok(); + + if let Err(ref err) = res { + match err.code() { + Dxgi::DXGI_ERROR_SDK_COMPONENT_MISSING => return Ok(None), + _ => {} + } + } + + res.into_device_result("debug_interface1")?; + + result__.ok_or(crate::DeviceError::Unexpected).map(Some) + } + + /// Will error with crate::DeviceError::Unexpected if DXGI 1.4 is not available. + pub fn create_factory4( + &self, + factory_flags: Dxgi::DXGI_CREATE_FACTORY_FLAGS, + ) -> Result { + // Calls windows::Win32::Graphics::Dxgi::CreateDXGIFactory2 on dxgi.dll + type Fun = extern "system" fn( + flags: Dxgi::DXGI_CREATE_FACTORY_FLAGS, + riid: *const windows_core::GUID, + ppfactory: *mut *mut ffi::c_void, + ) -> windows_core::HRESULT; + let func: libloading::Symbol = + unsafe { self.lib.get(c"CreateDXGIFactory2".to_bytes()) }?; + + let mut result__ = None; + + (func)( + factory_flags, + &Dxgi::IDXGIFactory4::IID, + <*mut _>::cast(&mut result__), + ) + .ok() + .into_device_result("create_factory4")?; + + result__.ok_or(crate::DeviceError::Unexpected) + } + + /// Will error with crate::DeviceError::Unexpected if DXGI 1.3 is not available. + pub fn create_factory_media(&self) -> Result { + // Calls windows::Win32::Graphics::Dxgi::CreateDXGIFactory1 on dxgi.dll + type Fun = extern "system" fn( + riid: *const windows_core::GUID, + ppfactory: *mut *mut ffi::c_void, + ) -> windows_core::HRESULT; + let func: libloading::Symbol = + unsafe { self.lib.get(c"CreateDXGIFactory1".to_bytes()) }?; + + let mut result__ = None; + + // https://learn.microsoft.com/en-us/windows/win32/api/dxgi1_3/nn-dxgi1_3-idxgifactorymedia + (func)(&Dxgi::IDXGIFactoryMedia::IID, <*mut _>::cast(&mut result__)) + .ok() + .into_device_result("create_factory_media")?; + + result__.ok_or(crate::DeviceError::Unexpected) + } +} + +/// Create a temporary "owned" copy inside a [`mem::ManuallyDrop`] without increasing the refcount or +/// moving away the source variable. +/// +/// This is a common pattern when needing to pass interface pointers ("borrows") into Windows +/// structs. Moving/cloning ownership is impossible/inconvenient because: +/// +/// - The caller does _not_ assume ownership (and decrement the refcount at a later time); +/// - Unnecessarily increasing and decrementing the refcount; +/// - [`Drop`] destructors cannot run inside `union` structures (when the created structure is +/// implicitly dropped after a call). +/// +/// See also and +/// . +/// +/// # Safety +/// Performs a [`mem::transmute_copy()`] on a refcounted [`Interface`] type. The returned +/// [`mem::ManuallyDrop`] should _not_ be dropped. +pub unsafe fn borrow_interface_temporarily(src: &I) -> mem::ManuallyDrop> { + unsafe { mem::transmute_copy(src) } +} + +/// See [`borrow_interface_temporarily()`] +pub unsafe fn borrow_optional_interface_temporarily( + src: &Option, +) -> mem::ManuallyDrop> { + unsafe { mem::transmute_copy(src) } +} + +struct D3DBlob(Direct3D::ID3DBlob); + +impl Deref for D3DBlob { + type Target = Direct3D::ID3DBlob; + + fn deref(&self) -> &Self::Target { + &self.0 + } +} + +impl D3DBlob { + unsafe fn as_slice(&self) -> &[u8] { + unsafe { core::slice::from_raw_parts(self.GetBufferPointer().cast(), self.GetBufferSize()) } + } + + unsafe fn as_c_str(&self) -> Result<&ffi::CStr, ffi::FromBytesUntilNulError> { + ffi::CStr::from_bytes_until_nul(unsafe { self.as_slice() }) + } +} + +#[derive(Clone, Debug)] +pub struct Api; + +impl crate::Api for Api { + const VARIANT: wgt::Backend = wgt::Backend::Dx12; + + type Instance = Instance; + type Surface = Surface; + type Adapter = Adapter; + type Device = Device; + + type Queue = Queue; + type CommandEncoder = CommandEncoder; + type CommandBuffer = CommandBuffer; + + type Buffer = Buffer; + type Texture = Texture; + type SurfaceTexture = Texture; + type TextureView = TextureView; + type Sampler = Sampler; + type QuerySet = QuerySet; + type Fence = Fence; + + type BindGroupLayout = BindGroupLayout; + type BindGroup = BindGroup; + type PipelineLayout = PipelineLayout; + type ShaderModule = ShaderModule; + type RenderPipeline = RenderPipeline; + type ComputePipeline = ComputePipeline; + type PipelineCache = PipelineCache; + + type AccelerationStructure = AccelerationStructure; +} + +crate::impl_dyn_resource!( + Adapter, + AccelerationStructure, + BindGroup, + BindGroupLayout, + Buffer, + CommandBuffer, + CommandEncoder, + ComputePipeline, + Device, + Fence, + Instance, + PipelineCache, + PipelineLayout, + QuerySet, + Queue, + RenderPipeline, + Sampler, + ShaderModule, + Surface, + Texture, + TextureView +); + +// Limited by D3D12's root signature size of 64. Each element takes 1 or 2 entries. +const MAX_ROOT_ELEMENTS: usize = 64; +const ZERO_BUFFER_SIZE: wgt::BufferAddress = 256 << 10; + +pub struct Instance { + factory: DxgiFactory, + factory_media: Option, + // `device_factory` must be dropped before `library` because the COM + // object's Release call goes through the d3d12.dll vtable. If + // `library` (which unloads d3d12.dll) is dropped first the Release + // segfaults. + device_factory: Arc, + library: Arc, + dcomp_lib: Arc, + supports_allow_tearing: bool, + presentation_system: wgt::Dx12SwapchainKind, + _lib_dxgi: DxgiLib, + flags: wgt::InstanceFlags, + memory_budget_thresholds: wgt::MemoryBudgetThresholds, + compiler_container: Arc, + options: wgt::Dx12BackendOptions, + telemetry: Option, +} + +impl Instance { + /// Get the raw DXGI factory associated with this instance. + pub unsafe fn raw_factory4(&self) -> &Dxgi::IDXGIFactory4 { + self.factory.deref() + } + + pub unsafe fn create_surface_from_visual(&self, visual: *mut ffi::c_void) -> Surface { + let visual = unsafe { DirectComposition::IDCompositionVisual::from_raw_borrowed(&visual) } + .expect("COM pointer should not be NULL"); + Surface { + factory: self.factory.clone(), + factory_media: self.factory_media.clone(), + target: SurfaceTarget::Visual(visual.to_owned()), + supports_allow_tearing: self.supports_allow_tearing, + swap_chain: RwLock::new(None), + options: self.options.clone(), + } + } + + pub unsafe fn create_surface_from_surface_handle( + &self, + surface_handle: *mut ffi::c_void, + ) -> Surface { + // TODO: We're not given ownership, so we shouldn't call HANDLE::free(). This puts an extra burden on the caller to keep it alive. + // https://learn.microsoft.com/en-us/windows/win32/api/handleapi/nf-handleapi-duplicatehandle could help us, even though DirectComposition is not in the list? + // Or we make all these types owned, require an ownership transition, and replace SurfaceTargetUnsafe with SurfaceTarget. + let surface_handle = Foundation::HANDLE(surface_handle); + Surface { + factory: self.factory.clone(), + factory_media: self.factory_media.clone(), + target: SurfaceTarget::SurfaceHandle(surface_handle), + supports_allow_tearing: self.supports_allow_tearing, + swap_chain: RwLock::new(None), + options: self.options.clone(), + } + } + + pub unsafe fn create_surface_from_swap_chain_panel( + &self, + swap_chain_panel: *mut ffi::c_void, + ) -> Surface { + let swap_chain_panel = + unsafe { types::ISwapChainPanelNative::from_raw_borrowed(&swap_chain_panel) } + .expect("COM pointer should not be NULL"); + Surface { + factory: self.factory.clone(), + factory_media: self.factory_media.clone(), + target: SurfaceTarget::SwapChainPanel(swap_chain_panel.to_owned()), + supports_allow_tearing: self.supports_allow_tearing, + swap_chain: RwLock::new(None), + options: self.options.clone(), + } + } +} + +unsafe impl Send for Instance {} +unsafe impl Sync for Instance {} + +struct SwapChain { + // TODO: Drop order frees the SWC before the raw image pointers...? + raw: Dxgi::IDXGISwapChain3, + // need to associate raw image pointers with the swapchain so they can be properly released + // when the swapchain is destroyed + resources: Vec, + /// Handle is freed in [`Self::release_resources()`] + waitable: Option, + acquired_count: usize, + present_mode: wgt::PresentMode, + format: wgt::TextureFormat, + size: wgt::Extent3d, +} + +enum SurfaceTarget { + /// Borrowed, lifetime externally managed + WndHandle(Foundation::HWND), + /// `handle` is borrowed, lifetime externally managed + VisualFromWndHandle { + handle: Foundation::HWND, + dcomp_state: Mutex, + }, + Visual(DirectComposition::IDCompositionVisual), + /// Borrowed, lifetime externally managed + SurfaceHandle(Foundation::HANDLE), + SwapChainPanel(types::ISwapChainPanelNative), +} + +pub struct Surface { + factory: DxgiFactory, + factory_media: Option, + target: SurfaceTarget, + supports_allow_tearing: bool, + swap_chain: RwLock>, + options: wgt::Dx12BackendOptions, +} + +unsafe impl Send for Surface {} +unsafe impl Sync for Surface {} + +impl Surface { + pub fn swap_chain(&self) -> Option { + Some(self.swap_chain.read().as_ref()?.raw.clone()) + } + + /// Returns the waitable handle associated with this swap chain, if any. + /// Handle is only valid while the swap chain is alive. + pub unsafe fn waitable_handle(&self) -> Option { + self.swap_chain.read().as_ref()?.waitable + } +} + +#[derive(Debug, Clone, Copy)] +enum MemoryArchitecture { + Unified { + #[allow(unused)] + cache_coherent: bool, + }, + NonUnified, +} + +#[derive(Debug, Clone, Copy)] +struct PrivateCapabilities { + instance_flags: wgt::InstanceFlags, + workarounds: Workarounds, + #[allow(unused)] + heterogeneous_resource_heaps: bool, + memory_architecture: MemoryArchitecture, + heap_create_not_zeroed: bool, + casting_fully_typed_format_supported: bool, + suballocation_supported: bool, + shader_model: naga::back::hlsl::ShaderModel, + max_sampler_descriptor_heap_size: u32, + unrestricted_buffer_texture_copy_pitch_supported: bool, +} + +impl PrivateCapabilities { + fn texture_data_placement_alignment(&self) -> u64 { + if self.unrestricted_buffer_texture_copy_pitch_supported { + 4 + } else { + D3D12_TEXTURE_DATA_PLACEMENT_ALIGNMENT.into() + } + } +} + +#[derive(Default, Debug, Copy, Clone)] +struct Workarounds { + // On WARP 1.0.13+, debug information in shaders in certain situations causes the device + // to hang. https://github.com/gfx-rs/wgpu/issues/8368 + avoid_shader_debug_info: bool, +} + +pub struct Adapter { + raw: DxgiAdapter, + device: Direct3D12::ID3D12Device, + library: Arc, + dcomp_lib: Arc, + private_caps: PrivateCapabilities, + presentation_timer: auxil::dxgi::time::PresentationTimer, + memory_budget_thresholds: wgt::MemoryBudgetThresholds, + compiler_container: Arc, + options: wgt::Dx12BackendOptions, +} + +unsafe impl Send for Adapter {} +unsafe impl Sync for Adapter {} + +impl Adapter { + pub fn as_raw(&self) -> &Dxgi::IDXGIAdapter3 { + &self.raw + } +} + +struct Event(pub Foundation::HANDLE); +impl Event { + pub fn create(manual_reset: bool, initial_state: bool) -> Result { + Ok(Self( + unsafe { Threading::CreateEventA(None, manual_reset, initial_state, None) } + .into_device_result("CreateEventA")?, + )) + } +} + +impl Drop for Event { + fn drop(&mut self) { + unsafe { Foundation::HANDLE::free(&mut self.0) } + } +} + +/// Helper structure for waiting for GPU. +struct Idler { + fence: Direct3D12::ID3D12Fence, +} + +#[derive(Debug, Clone)] +struct CommandSignatures { + draw: Direct3D12::ID3D12CommandSignature, + draw_indexed: Direct3D12::ID3D12CommandSignature, + draw_mesh: Option, + dispatch: Direct3D12::ID3D12CommandSignature, +} + +struct DeviceShared { + adapter: DxgiAdapter, + zero_buffer: Direct3D12::ID3D12Resource, + cmd_signatures: CommandSignatures, + heap_views: descriptor::GeneralHeap, + sampler_heap: sampler::SamplerHeap, + private_caps: PrivateCapabilities, +} + +unsafe impl Send for DeviceShared {} +unsafe impl Sync for DeviceShared {} + +pub struct Device { + raw: Direct3D12::ID3D12Device, + present_queue: Direct3D12::ID3D12CommandQueue, + idler: Idler, + features: wgt::Features, + shared: Arc, + options: wgt::Dx12BackendOptions, + // CPU only pools + rtv_pool: Arc>, + dsv_pool: Mutex, + srv_uav_pool: Mutex, + // library + library: Arc, + dcomp_lib: Arc, + #[cfg(feature = "renderdoc")] + render_doc: auxil::renderdoc::RenderDoc, + null_rtv_handle: descriptor::Handle, + mem_allocator: Allocator, + compiler_container: Arc, + shader_cache: Mutex, + counters: Arc, +} + +impl Drop for Device { + fn drop(&mut self) { + self.rtv_pool.lock().free_handle(self.null_rtv_handle); + if self + .shared + .private_caps + .instance_flags + .contains(wgt::InstanceFlags::VALIDATION) + { + auxil::dxgi::exception::unregister_exception_handler(); + } + } +} + +unsafe impl Send for Device {} +unsafe impl Sync for Device {} + +pub struct Queue { + raw: Direct3D12::ID3D12CommandQueue, + temp_lists: Mutex>>, +} + +impl Queue { + pub fn as_raw(&self) -> &Direct3D12::ID3D12CommandQueue { + &self.raw + } +} + +unsafe impl Send for Queue {} +unsafe impl Sync for Queue {} + +#[derive(Default)] +struct Temp { + marker: Vec, + barriers: Vec, +} + +impl Temp { + fn clear(&mut self) { + self.marker.clear(); + self.barriers.clear(); + } +} + +struct PassResolve { + src: (Direct3D12::ID3D12Resource, u32), + dst: (Direct3D12::ID3D12Resource, u32), + format: Dxgi::Common::DXGI_FORMAT, +} + +#[derive(Clone, Copy, Debug)] +enum RootElement { + Empty, + Constant, + SpecialConstantBuffer { + /// The first vertex in an indirect draw call, _or_ the `x` of a compute dispatch. + first_vertex: i32, + /// The first instance in an indirect draw call, _or_ the `y` of a compute dispatch. + first_instance: u32, + /// Unused in an indirect draw call, _or_ the `z` of a compute dispatch. + other: u32, + }, + /// Descriptor table. + Table(Direct3D12::D3D12_GPU_DESCRIPTOR_HANDLE), + /// Descriptor for an uniform buffer that has dynamic offset. + DynamicUniformBuffer { + address: Direct3D12::D3D12_GPU_DESCRIPTOR_HANDLE, + }, + /// Descriptor table referring to the entire sampler heap. + SamplerHeap, + /// Root constants for dynamic offsets. + /// + /// start..end is the range of values in [`PassState::dynamic_storage_buffer_offsets`] + /// that will be used to update the root constants. + DynamicOffsetsBuffer { + start: usize, + end: usize, + }, +} + +#[derive(Clone, Copy)] +enum PassKind { + Render, + Compute, + Transfer, +} + +struct PassState { + has_label: bool, + resolves: ArrayVec, + layout: PipelineLayoutShared, + root_elements: [RootElement; MAX_ROOT_ELEMENTS], + constant_data: [u32; MAX_ROOT_ELEMENTS], + dynamic_storage_buffer_offsets: Vec, + dirty_root_elements: u64, + vertex_buffers: [Direct3D12::D3D12_VERTEX_BUFFER_VIEW; crate::MAX_VERTEX_BUFFERS], + dirty_vertex_buffers: usize, + kind: PassKind, +} + +#[test] +fn test_dirty_mask() { + assert_eq!(MAX_ROOT_ELEMENTS, u64::BITS as usize); +} + +impl PassState { + fn new() -> Self { + PassState { + has_label: false, + resolves: ArrayVec::new(), + layout: PipelineLayoutShared { + signature: None, + total_root_elements: 0, + special_constants: None, + root_constant_info: None, + sampler_heap_root_index: None, + }, + root_elements: [RootElement::Empty; MAX_ROOT_ELEMENTS], + constant_data: [0; MAX_ROOT_ELEMENTS], + dynamic_storage_buffer_offsets: Vec::new(), + dirty_root_elements: 0, + vertex_buffers: [Default::default(); crate::MAX_VERTEX_BUFFERS], + dirty_vertex_buffers: 0, + kind: PassKind::Transfer, + } + } + + fn clear(&mut self) { + // careful about heap allocations! + *self = Self::new(); + } +} + +pub struct CommandEncoder { + allocator: Direct3D12::ID3D12CommandAllocator, + device: Direct3D12::ID3D12Device, + shared: Arc, + mem_allocator: Allocator, + + rtv_pool: Arc>, + temp_rtv_handles: Vec, + + intermediate_copy_bufs: Vec, + + null_rtv_handle: descriptor::Handle, + list: Option, + free_lists: Vec, + pass: PassState, + temp: Temp, + + /// If set, the end of the next render/compute pass will write a timestamp at + /// the given pool & location. + end_of_pass_timer_query: Option<(Direct3D12::ID3D12QueryHeap, u32)>, + + counters: Arc, +} + +unsafe impl Send for CommandEncoder {} +unsafe impl Sync for CommandEncoder {} + +impl fmt::Debug for CommandEncoder { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("CommandEncoder") + .field("allocator", &self.allocator) + .field("device", &self.allocator) + .finish() + } +} + +#[derive(Debug)] +pub struct CommandBuffer { + raw: Direct3D12::ID3D12GraphicsCommandList, +} + +impl crate::DynCommandBuffer for CommandBuffer {} + +unsafe impl Send for CommandBuffer {} +unsafe impl Sync for CommandBuffer {} + +#[derive(Debug)] +pub struct Buffer { + resource: Direct3D12::ID3D12Resource, + // While the allocation also has _a_ size, it may not + // be the same as the original size of the buffer, + // as the allocation size varies for assorted reasons. + size: wgt::BufferAddress, + allocation: suballocation::Allocation, +} + +unsafe impl Send for Buffer {} +unsafe impl Sync for Buffer {} + +impl crate::DynBuffer for Buffer {} + +impl crate::BufferBinding<'_, Buffer> { + fn resolve_size(&self) -> wgt::BufferAddress { + match self.size { + Some(size) => size.get(), + None => self.buffer.size - self.offset, + } + } + + // TODO: Return GPU handle directly? + fn resolve_address(&self) -> wgt::BufferAddress { + (unsafe { self.buffer.resource.GetGPUVirtualAddress() }) + self.offset + } +} + +#[derive(Debug)] +pub struct Texture { + resource: Direct3D12::ID3D12Resource, + format: wgt::TextureFormat, + dimension: wgt::TextureDimension, + size: wgt::Extent3d, + mip_level_count: u32, + sample_count: u32, + allocation: suballocation::Allocation, +} + +impl Texture { + pub unsafe fn raw_resource(&self) -> &Direct3D12::ID3D12Resource { + &self.resource + } +} + +impl crate::DynTexture for Texture {} +impl crate::DynSurfaceTexture for Texture {} + +impl core::borrow::Borrow for Texture { + fn borrow(&self) -> &dyn crate::DynTexture { + self + } +} + +unsafe impl Send for Texture {} +unsafe impl Sync for Texture {} + +impl Texture { + fn array_layer_count(&self) -> u32 { + match self.dimension { + wgt::TextureDimension::D1 | wgt::TextureDimension::D3 => 1, + wgt::TextureDimension::D2 => self.size.depth_or_array_layers, + } + } + + /// see + fn calc_subresource(&self, mip_level: u32, array_layer: u32, plane: u32) -> u32 { + mip_level + (array_layer + plane * self.array_layer_count()) * self.mip_level_count + } + + fn calc_subresource_for_copy(&self, base: &crate::TextureCopyBase) -> u32 { + let plane = match base.aspect { + crate::FormatAspects::COLOR + | crate::FormatAspects::DEPTH + | crate::FormatAspects::PLANE_0 => 0, + crate::FormatAspects::STENCIL | crate::FormatAspects::PLANE_1 => 1, + crate::FormatAspects::PLANE_2 => 2, + _ => unreachable!(), + }; + self.calc_subresource(base.mip_level, base.array_layer, plane) + } +} + +#[derive(Debug)] +pub struct TextureView { + raw_format: Dxgi::Common::DXGI_FORMAT, + aspects: crate::FormatAspects, + dimension: wgt::TextureViewDimension, + texture: Direct3D12::ID3D12Resource, + subresource_index: u32, + mip_slice: u32, + handle_srv: Option, + handle_uav: Option, + handle_rtv: Option, + handle_dsv_ro: Option, + handle_dsv_rw: Option, +} + +impl crate::DynTextureView for TextureView {} + +unsafe impl Send for TextureView {} +unsafe impl Sync for TextureView {} + +#[derive(Debug)] +pub struct Sampler { + index: sampler::SamplerIndex, + desc: Direct3D12::D3D12_SAMPLER_DESC, +} + +impl crate::DynSampler for Sampler {} + +unsafe impl Send for Sampler {} +unsafe impl Sync for Sampler {} + +#[derive(Debug)] +pub struct QuerySet { + raw: Direct3D12::ID3D12QueryHeap, + raw_ty: Direct3D12::D3D12_QUERY_TYPE, +} + +impl crate::DynQuerySet for QuerySet {} + +unsafe impl Send for QuerySet {} +unsafe impl Sync for QuerySet {} + +#[derive(Debug)] +pub struct Fence { + raw: Direct3D12::ID3D12Fence, +} + +impl crate::DynFence for Fence {} + +unsafe impl Send for Fence {} +unsafe impl Sync for Fence {} + +impl Fence { + pub fn raw_fence(&self) -> &Direct3D12::ID3D12Fence { + &self.raw + } +} + +#[derive(Debug)] +pub struct BindGroupLayout { + /// Sorted list of entries. + entries: Vec, + cpu_heap_views: Option, + copy_counts: Vec, // all 1's +} + +impl crate::DynBindGroupLayout for BindGroupLayout {} + +#[derive(Debug, Clone, Copy)] +enum DynamicBuffer { + Uniform(Direct3D12::D3D12_GPU_DESCRIPTOR_HANDLE), + Storage, +} + +#[derive(Debug)] +struct SamplerIndexBuffer { + buffer: Direct3D12::ID3D12Resource, + allocation: suballocation::Allocation, +} + +#[derive(Debug)] +pub struct BindGroup { + handle_views: Option, + sampler_index_buffer: Option, + dynamic_buffers: Vec, +} + +impl crate::DynBindGroup for BindGroup {} + +bitflags::bitflags! { + #[derive(Debug, Copy, Clone, PartialEq, Eq, Hash)] + struct TableTypes: u8 { + const SRV_CBV_UAV = 1 << 0; + const SAMPLERS = 1 << 1; + } +} + +// Element (also known as parameter) index into the root signature. +type RootIndex = u32; + +#[derive(Debug)] +struct BindGroupInfo { + base_root_index: RootIndex, + tables: TableTypes, + dynamic_storage_buffer_offsets: Option, +} + +#[derive(Debug, Clone)] +struct RootConstantInfo { + root_index: RootIndex, + range: core::ops::Range, +} + +#[derive(Debug, Clone)] +struct DynamicStorageBufferOffsets { + root_index: RootIndex, + range: core::ops::Range, +} + +#[derive(Debug, Clone)] +struct PipelineLayoutShared { + signature: Option, + total_root_elements: RootIndex, + special_constants: Option, + root_constant_info: Option, + sampler_heap_root_index: Option, +} + +unsafe impl Send for PipelineLayoutShared {} +unsafe impl Sync for PipelineLayoutShared {} + +#[derive(Debug, Clone)] +struct PipelineLayoutSpecialConstants { + root_index: RootIndex, + indirect_cmd_signatures: Option, +} + +unsafe impl Send for PipelineLayoutSpecialConstants {} +unsafe impl Sync for PipelineLayoutSpecialConstants {} + +#[derive(Debug)] +pub struct PipelineLayout { + shared: PipelineLayoutShared, + // Storing for each associated bind group, which tables we created + // in the root signature. This is required for binding descriptor sets. + bind_group_infos: [Option; crate::MAX_BIND_GROUPS], + naga_options: naga::back::hlsl::Options, +} + +impl crate::DynPipelineLayout for PipelineLayout {} + +#[derive(Debug)] +pub struct ShaderModule { + source: ShaderModuleSource, + raw_name: Option, + runtime_checks: wgt::ShaderRuntimeChecks, +} + +impl crate::DynShaderModule for ShaderModule {} + +#[derive(Default)] +pub struct ShaderCache { + nr_of_shaders_compiled: u32, + entries: HashMap, +} + +#[derive(PartialEq, Eq, Hash)] +pub(super) struct ShaderCacheKey { + source: String, + entry_point: String, + stage: naga::ShaderStage, + shader_model: naga::back::hlsl::ShaderModel, +} + +pub(super) struct ShaderCacheValue { + /// This is the value of [`ShaderCache::nr_of_shaders_compiled`] + /// at the time the cache entry was last used. + last_used: u32, + shader: CompiledShader, +} + +#[derive(Clone)] +pub(super) enum CompiledShader { + Dxc(Direct3D::Dxc::IDxcBlob), + Fxc(Direct3D::ID3DBlob), + Precompiled(Vec), +} + +impl CompiledShader { + fn create_native_shader(&self) -> Direct3D12::D3D12_SHADER_BYTECODE { + match self { + CompiledShader::Dxc(shader) => Direct3D12::D3D12_SHADER_BYTECODE { + pShaderBytecode: unsafe { shader.GetBufferPointer() }, + BytecodeLength: unsafe { shader.GetBufferSize() }, + }, + CompiledShader::Fxc(shader) => Direct3D12::D3D12_SHADER_BYTECODE { + pShaderBytecode: unsafe { shader.GetBufferPointer() }, + BytecodeLength: unsafe { shader.GetBufferSize() }, + }, + CompiledShader::Precompiled(shader) => Direct3D12::D3D12_SHADER_BYTECODE { + pShaderBytecode: shader.as_ptr().cast(), + BytecodeLength: shader.len(), + }, + } + } +} + +#[derive(Debug)] +pub struct RenderPipeline { + raw: Direct3D12::ID3D12PipelineState, + layout: PipelineLayoutShared, + topology: Direct3D::D3D_PRIMITIVE_TOPOLOGY, + vertex_strides: [Option; crate::MAX_VERTEX_BUFFERS], +} + +impl crate::DynRenderPipeline for RenderPipeline {} + +unsafe impl Send for RenderPipeline {} +unsafe impl Sync for RenderPipeline {} + +#[derive(Debug)] +pub struct ComputePipeline { + raw: Direct3D12::ID3D12PipelineState, + layout: PipelineLayoutShared, +} + +impl crate::DynComputePipeline for ComputePipeline {} + +unsafe impl Send for ComputePipeline {} +unsafe impl Sync for ComputePipeline {} + +#[derive(Debug)] +pub struct PipelineCache; + +impl crate::DynPipelineCache for PipelineCache {} + +#[derive(Debug)] +pub struct AccelerationStructure { + resource: Direct3D12::ID3D12Resource, + allocation: suballocation::Allocation, +} + +impl crate::DynAccelerationStructure for AccelerationStructure {} + +impl SwapChain { + unsafe fn release_resources(mut self) -> Dxgi::IDXGISwapChain3 { + if let Some(mut waitable) = self.waitable.take() { + unsafe { Foundation::HANDLE::free(&mut waitable) }; + } + self.raw + } + + unsafe fn wait( + &mut self, + timeout: Option, + ) -> Result { + let timeout_ms = match timeout { + Some(duration) => duration.as_millis() as u32, + None => Threading::INFINITE, + }; + + if let Some(waitable) = self.waitable { + match unsafe { Threading::WaitForSingleObject(waitable, timeout_ms) } { + Foundation::WAIT_ABANDONED | Foundation::WAIT_FAILED => { + Err(crate::SurfaceError::Lost) + } + Foundation::WAIT_OBJECT_0 => Ok(true), + Foundation::WAIT_TIMEOUT => Ok(false), + other => { + log::error!("Unexpected wait status: 0x{other:x?}"); + Err(crate::SurfaceError::Lost) + } + } + } else { + Ok(true) + } + } +} + +impl crate::Surface for Surface { + type A = Api; + + unsafe fn configure( + &self, + device: &Device, + config: &crate::SurfaceConfiguration, + ) -> Result<(), crate::SurfaceError> { + let mut flags = Dxgi::DXGI_SWAP_CHAIN_FLAG_FRAME_LATENCY_WAITABLE_OBJECT; + // We always set ALLOW_TEARING on the swapchain no matter + // what kind of swapchain we want because ResizeBuffers + // cannot change the swapchain's ALLOW_TEARING flag. + // + // This does not change the behavior of the swapchain, just + // allow present calls to use tearing. + if self.supports_allow_tearing { + flags |= Dxgi::DXGI_SWAP_CHAIN_FLAG_ALLOW_TEARING; + } + + // While `configure`s contract ensures that no work on the GPU's main queues + // are in flight, we still need to wait for the present queue to be idle. + unsafe { device.wait_for_present_queue_idle() }?; + + let non_srgb_format = auxil::dxgi::conv::map_texture_format_nosrgb(config.format); + + // The range for `SetMaximumFrameLatency` is 1-16 so the maximum latency requested should be 15 because we add 1. + // https://learn.microsoft.com/en-us/windows/win32/api/dxgi/nf-dxgi-idxgidevice1-setmaximumframelatency + debug_assert!(config.maximum_frame_latency <= 15); + + // Nvidia recommends to use 1-2 more buffers than the maximum latency + // https://developer.nvidia.com/blog/advanced-api-performance-swap-chains/ + // For high latency extra buffers seems excessive, so go with a minimum of 3 and beyond that add 1. + let swap_chain_buffer = (config.maximum_frame_latency + 1).min(16); + + let swap_chain = match self.swap_chain.write().take() { + //Note: this path doesn't properly re-initialize all of the things + Some(sc) => { + let raw = unsafe { sc.release_resources() }; + let result = unsafe { + raw.ResizeBuffers( + swap_chain_buffer, + config.extent.width, + config.extent.height, + non_srgb_format, + flags, + ) + }; + if let Err(err) = result { + log::error!("ResizeBuffers failed: {err}"); + return Err(crate::SurfaceError::Other("window is in use")); + } + raw + } + None => { + let desc = Dxgi::DXGI_SWAP_CHAIN_DESC1 { + AlphaMode: auxil::dxgi::conv::map_acomposite_alpha_mode( + config.composite_alpha_mode, + ), + Width: config.extent.width, + Height: config.extent.height, + Format: non_srgb_format, + Stereo: false.into(), + SampleDesc: Dxgi::Common::DXGI_SAMPLE_DESC { + Count: 1, + Quality: 0, + }, + BufferUsage: Dxgi::DXGI_USAGE_RENDER_TARGET_OUTPUT, + BufferCount: swap_chain_buffer, + Scaling: Dxgi::DXGI_SCALING_STRETCH, + SwapEffect: Dxgi::DXGI_SWAP_EFFECT_FLIP_DISCARD, + Flags: flags.0 as u32, + }; + let swap_chain1 = match self.target { + SurfaceTarget::Visual(_) + | SurfaceTarget::VisualFromWndHandle { .. } + | SurfaceTarget::SwapChainPanel(_) => { + profiling::scope!("IDXGIFactory2::CreateSwapChainForComposition"); + unsafe { + self.factory.CreateSwapChainForComposition( + &device.present_queue, + &desc, + None, + ) + } + } + SurfaceTarget::SurfaceHandle(handle) => { + profiling::scope!( + "IDXGIFactoryMedia::CreateSwapChainForCompositionSurfaceHandle" + ); + unsafe { + self.factory_media + .as_ref() + .ok_or(crate::SurfaceError::Other("IDXGIFactoryMedia not found"))? + .CreateSwapChainForCompositionSurfaceHandle( + &device.present_queue, + Some(handle), + &desc, + None, + ) + } + } + SurfaceTarget::WndHandle(hwnd) => { + profiling::scope!("IDXGIFactory2::CreateSwapChainForHwnd"); + unsafe { + self.factory.CreateSwapChainForHwnd( + &device.present_queue, + hwnd, + &desc, + None, + None, + ) + } + } + }; + + let swap_chain1 = swap_chain1.map_err(|err| { + log::error!("SwapChain creation error: {err}"); + crate::SurfaceError::Other("swapchain creation") + })?; + + match &self.target { + SurfaceTarget::WndHandle(_) | SurfaceTarget::SurfaceHandle(_) => {} + SurfaceTarget::VisualFromWndHandle { + handle, + dcomp_state, + } => { + let mut dcomp_state = dcomp_state.lock(); + let dcomp_state = + unsafe { dcomp_state.get_or_init(&device.dcomp_lib, handle) }?; + // Set the new swap chain as the content for the backing visual + // and commit the changes to the composition visual tree. + { + profiling::scope!("IDCompositionVisual::SetContent"); + unsafe { dcomp_state.visual.SetContent(&swap_chain1) }.map_err( + |err| { + log::error!("IDCompositionVisual::SetContent failed: {err}"); + crate::SurfaceError::Other("IDCompositionVisual::SetContent") + }, + )?; + } + + // Commit the changes to the composition device. + { + profiling::scope!("IDCompositionDevice::Commit"); + unsafe { dcomp_state.device.Commit() }.map_err(|err| { + log::error!("IDCompositionDevice::Commit failed: {err}"); + crate::SurfaceError::Other("IDCompositionDevice::Commit") + })?; + } + } + SurfaceTarget::Visual(visual) => { + if let Err(err) = unsafe { visual.SetContent(&swap_chain1) } { + log::error!("Unable to SetContent: {err}"); + return Err(crate::SurfaceError::Other( + "IDCompositionVisual::SetContent", + )); + } + } + SurfaceTarget::SwapChainPanel(swap_chain_panel) => { + if let Err(err) = unsafe { swap_chain_panel.SetSwapChain(&swap_chain1) } { + log::error!("Unable to SetSwapChain: {err}"); + return Err(crate::SurfaceError::Other( + "ISwapChainPanelNative::SetSwapChain", + )); + } + } + } + + swap_chain1.cast::().map_err(|err| { + log::error!("Unable to cast swapchain: {err}"); + crate::SurfaceError::Other("swapchain cast to version 3") + })? + } + }; + + match self.target { + SurfaceTarget::WndHandle(wnd_handle) => { + // Disable automatic Alt+Enter handling by DXGI. + unsafe { + self.factory.MakeWindowAssociation( + wnd_handle, + Dxgi::DXGI_MWA_NO_WINDOW_CHANGES | Dxgi::DXGI_MWA_NO_ALT_ENTER, + ) + } + .into_device_result("MakeWindowAssociation")?; + } + SurfaceTarget::Visual(_) + | SurfaceTarget::VisualFromWndHandle { .. } + | SurfaceTarget::SurfaceHandle(_) + | SurfaceTarget::SwapChainPanel(_) => {} + } + + unsafe { swap_chain.SetMaximumFrameLatency(config.maximum_frame_latency) } + .into_device_result("SetMaximumFrameLatency")?; + + let waitable = match device.options.latency_waitable_object { + wgt::Dx12UseFrameLatencyWaitableObject::None => None, + wgt::Dx12UseFrameLatencyWaitableObject::Wait + | wgt::Dx12UseFrameLatencyWaitableObject::DontWait => { + Some(unsafe { swap_chain.GetFrameLatencyWaitableObject() }) + } + }; + + let mut resources = Vec::with_capacity(swap_chain_buffer as usize); + for i in 0..swap_chain_buffer { + let resource = unsafe { swap_chain.GetBuffer(i) } + .into_device_result("Failed to get swapchain buffer")?; + resources.push(resource); + } + + let mut swapchain = self.swap_chain.write(); + *swapchain = Some(SwapChain { + raw: swap_chain, + resources, + waitable, + acquired_count: 0, + present_mode: config.present_mode, + format: config.format, + size: config.extent, + }); + + Ok(()) + } + + unsafe fn unconfigure(&self, device: &Device) { + if let Some(sc) = self.swap_chain.write().take() { + unsafe { + // While `unconfigure`s contract ensures that no work on the GPU's main queues + // are in flight, we still need to wait for the present queue to be idle. + + // The major failure mode of this function is device loss, + // which if we have lost the device, we should just continue + // cleaning up, without error. + let _ = device.wait_for_present_queue_idle(); + + let _raw = sc.release_resources(); + } + } + } + + unsafe fn acquire_texture( + &self, + timeout: Option, + _fence: &Fence, + ) -> Result, crate::SurfaceError> { + let mut swapchain = self.swap_chain.write(); + let sc = swapchain.as_mut().unwrap(); + + match self.options.latency_waitable_object { + wgt::Dx12UseFrameLatencyWaitableObject::None + | wgt::Dx12UseFrameLatencyWaitableObject::DontWait => {} + wgt::Dx12UseFrameLatencyWaitableObject::Wait => { + unsafe { sc.wait(timeout) }?; + } + } + + let base_index = unsafe { sc.raw.GetCurrentBackBufferIndex() } as usize; + let index = (base_index + sc.acquired_count) % sc.resources.len(); + sc.acquired_count += 1; + + let texture = Texture { + resource: sc.resources[index].clone(), + format: sc.format, + dimension: wgt::TextureDimension::D2, + size: sc.size, + mip_level_count: 1, + sample_count: 1, + allocation: suballocation::Allocation::none( + suballocation::AllocationType::Texture, + sc.format.theoretical_memory_footprint(sc.size), + ), + }; + Ok(crate::AcquiredSurfaceTexture { + texture, + suboptimal: false, + }) + } + unsafe fn discard_texture(&self, _texture: Texture) { + let mut swapchain = self.swap_chain.write(); + let sc = swapchain.as_mut().unwrap(); + sc.acquired_count -= 1; + } +} + +impl crate::Queue for Queue { + type A = Api; + + unsafe fn submit( + &self, + command_buffers: &[&CommandBuffer], + _surface_textures: &[&Texture], + (signal_fence, signal_value): (&mut Fence, crate::FenceValue), + ) -> Result<(), crate::DeviceError> { + let mut temp_lists = self.temp_lists.lock(); + temp_lists.clear(); + for cmd_buf in command_buffers { + temp_lists.push(Some(cmd_buf.raw.clone().into())); + } + + { + profiling::scope!("ID3D12CommandQueue::ExecuteCommandLists"); + unsafe { self.raw.ExecuteCommandLists(&temp_lists) } + } + + unsafe { self.raw.Signal(&signal_fence.raw, signal_value) } + .into_device_result("Signal fence")?; + + // Note the lack of synchronization here between the main Direct queue + // and the dedicated presentation queue. This is automatically handled + // by the D3D runtime by detecting uses of resources derived from the + // swapchain. This automatic detection is why you cannot use a swapchain + // as an UAV in D3D12. + + Ok(()) + } + unsafe fn present( + &self, + surface: &Surface, + _texture: Texture, + ) -> Result<(), crate::SurfaceError> { + let mut swapchain = surface.swap_chain.write(); + let sc = swapchain.as_mut().unwrap(); + sc.acquired_count -= 1; + + let (interval, flags) = match sc.present_mode { + // We only allow immediate if ALLOW_TEARING is valid. + wgt::PresentMode::Immediate => (0, Dxgi::DXGI_PRESENT_ALLOW_TEARING), + wgt::PresentMode::Mailbox => (0, Dxgi::DXGI_PRESENT::default()), + wgt::PresentMode::Fifo => (1, Dxgi::DXGI_PRESENT::default()), + m => unreachable!("Cannot make surface with present mode {m:?}"), + }; + + profiling::scope!("IDXGISwapchain3::Present"); + unsafe { sc.raw.Present(interval, flags) } + .ok() + .into_device_result("Present")?; + + Ok(()) + } + + unsafe fn get_timestamp_period(&self) -> f32 { + let frequency = unsafe { self.raw.GetTimestampFrequency() }.expect("GetTimestampFrequency"); + (1_000_000_000.0 / frequency as f64) as f32 + } +} +#[derive(Debug)] +pub struct DxilPassthroughShader { + pub shader: Vec, + pub num_workgroups: (u32, u32, u32), +} + +#[derive(Debug)] +pub struct HlslPassthroughShader { + pub shader: String, + pub num_workgroups: (u32, u32, u32), +} + +#[derive(Debug)] +pub enum ShaderModuleSource { + Naga(crate::NagaShader), + DxilPassthrough(DxilPassthroughShader), + HlslPassthrough(HlslPassthroughShader), +} + +#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +pub enum FeatureLevel { + _11_0, + _11_1, + _12_0, + _12_1, + _12_2, +} + +#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +pub enum ShaderModel { + _5_1, + _6_0, + _6_1, + _6_2, + _6_3, + _6_4, + _6_5, + _6_6, + _6_7, + _6_8, + _6_9, +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/pipeline_desc.rs b/third_party/wgpu-hal-29.0.4/src/dx12/pipeline_desc.rs new file mode 100644 index 0000000..9fcf566 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/pipeline_desc.rs @@ -0,0 +1,344 @@ +//! We try to use pipeline stream descriptors where possible, but this isn't allowed +//! on some older windows 10 versions. Therefore, we also must have some logic to +//! convert such descriptors to the "traditional" equivalent, +//! `D3D12_GRAPHICS_PIPELINE_STATE_DESC`. +//! +//! Stream descriptors allow extending the pipeline, enabling more advanced features, +//! including mesh shaders and multiview/view instancing. Using a stream descriptor +//! is like using a vulkan descriptor with a `pNext` chain. It doesn't have direct +//! benefits to all use cases, but allows new use cases. +//! +//! The code for pipeline stream descriptors is very complicated, and can have bad +//! consequences if it is written incorrectly. It has been isolated to this file for +//! that reason. + +use core::{ffi::c_void, mem::ManuallyDrop, ptr::NonNull}; + +use alloc::vec::Vec; +use windows::Win32::Graphics::Direct3D12::*; +use windows::Win32::Graphics::Dxgi::Common::*; +use windows_core::Interface; + +use crate::dx12::borrow_interface_temporarily; + +// Wrapper newtypes for various pipeline subobjects which +// use complicated or non-unique representations. + +#[repr(transparent)] +#[derive(Copy, Clone)] +// Option> is guaranteed to have the same representation as a raw pointer. +struct RootSignature(Option>); + +#[repr(transparent)] +#[derive(Copy, Clone)] +struct VertexShader(D3D12_SHADER_BYTECODE); +#[repr(transparent)] +#[derive(Copy, Clone)] +struct PixelShader(D3D12_SHADER_BYTECODE); + +#[repr(transparent)] +#[derive(Copy, Clone)] +struct MeshShader(D3D12_SHADER_BYTECODE); + +#[repr(transparent)] +#[derive(Copy, Clone)] +struct TaskShader(D3D12_SHADER_BYTECODE); + +#[repr(transparent)] +#[derive(Copy, Clone)] +struct SampleMask(u32); + +#[repr(transparent)] +#[derive(Copy, Clone)] +struct NodeMask(u32); + +/// Trait for types that can be used as subobjects in a pipeline state stream. +/// +/// Safety: +/// - The type must be the correct alignment and size for the subobject it represents. +/// - The type must map to exactly one `D3D12_PIPELINE_STATE_SUBOBJECT_TYPE` variant. +/// - The variant must correctly represent the type's role in the pipeline state stream. +/// - The type must be `Copy` to ensure safe duplication in the stream. +/// - The type must be valid to memcpy into the pipeline state stream. +unsafe trait RenderPipelineStreamObject: Copy { + const SUBOBJECT_TYPE: D3D12_PIPELINE_STATE_SUBOBJECT_TYPE; +} + +macro_rules! implement_stream_object { + (unsafe $ty:ty => $variant:expr) => { + unsafe impl RenderPipelineStreamObject for $ty { + const SUBOBJECT_TYPE: D3D12_PIPELINE_STATE_SUBOBJECT_TYPE = $variant; + } + }; +} + +implement_stream_object! { unsafe RootSignature => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_ROOT_SIGNATURE } +implement_stream_object! { unsafe VertexShader => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_VS } +implement_stream_object! { unsafe PixelShader => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_PS } +implement_stream_object! { unsafe MeshShader => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_MS } +implement_stream_object! { unsafe TaskShader => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_AS } +implement_stream_object! { unsafe D3D12_BLEND_DESC => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_BLEND } +implement_stream_object! { unsafe SampleMask => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_SAMPLE_MASK } +implement_stream_object! { unsafe D3D12_RASTERIZER_DESC => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_RASTERIZER } +implement_stream_object! { unsafe D3D12_DEPTH_STENCIL_DESC => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_DEPTH_STENCIL } +implement_stream_object! { unsafe D3D12_PRIMITIVE_TOPOLOGY_TYPE => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_PRIMITIVE_TOPOLOGY } +implement_stream_object! { unsafe D3D12_RT_FORMAT_ARRAY => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_RENDER_TARGET_FORMATS } +implement_stream_object! { unsafe DXGI_FORMAT => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_DEPTH_STENCIL_FORMAT } +implement_stream_object! { unsafe DXGI_SAMPLE_DESC => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_SAMPLE_DESC } +implement_stream_object! { unsafe NodeMask => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_NODE_MASK } +implement_stream_object! { unsafe D3D12_CACHED_PIPELINE_STATE => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_CACHED_PSO } +implement_stream_object! { unsafe D3D12_PIPELINE_STATE_FLAGS => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_FLAGS } +implement_stream_object! { unsafe D3D12_INPUT_LAYOUT_DESC => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_INPUT_LAYOUT } +implement_stream_object! { unsafe D3D12_INDEX_BUFFER_STRIP_CUT_VALUE => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_IB_STRIP_CUT_VALUE } +implement_stream_object! { unsafe D3D12_STREAM_OUTPUT_DESC => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_STREAM_OUTPUT } +implement_stream_object! { unsafe D3D12_VIEW_INSTANCING_DESC => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE_VIEW_INSTANCING } + +/// Implementaation of a pipeline state stream, which is a sequence of subobjects put into +/// a byte array according to some basic alignment rules. +/// +/// Each subobject must start on an 8 byte boundary. Each subobject contains a 32 bit +/// type identifier, followed by the actual subobject data, aligned as required by the +/// subobject's structure. +/// +/// See +/// for more information. +pub(super) struct RenderPipelineStateStream<'a> { + bytes: Vec, + _marker: core::marker::PhantomData<&'a ()>, +} + +impl<'a> RenderPipelineStateStream<'a> { + fn new() -> Self { + // Dynamic allocation is used here because the resulting stream can become very large. + // We pre-allocate the size based on an estimate of the size of the struct plus some extra space + // per member for tags and alignment padding. In practice this will always be too big, as not + // all members will be used. + let size_of_stream_desc = size_of::(); + let members = 20; // Approximate number of members we might push + let capacity = size_of_stream_desc + members * 8; // Extra space for tags and alignment + Self { + bytes: Vec::with_capacity(capacity), + _marker: core::marker::PhantomData, + } + } + + /// Align the internal byte buffer to the given alignment, + /// padding with zeros as necessary. + fn align_to(&mut self, alignment: usize) { + let aligned_length = self.bytes.len().next_multiple_of(alignment); + self.bytes.resize(aligned_length, 0); + } + + /// Adds a subobject to the pipeline state stream. + fn add_object(&mut self, object: T) { + // Ensure 8-byte alignment for the subobject start. + self.align_to(8); + + // Append the type tag (u32) + let tag: u32 = T::SUBOBJECT_TYPE.0 as u32; + self.bytes.extend_from_slice(&tag.to_ne_bytes()); + + // Align the data to its natural alignment. + self.align_to(align_of_val::(&object)); + + // Append the data itself, as raw bytes + let data_ptr: *const T = &object; + let data_u8_ptr: *const u8 = data_ptr.cast::(); + let data_size = size_of_val::(&object); + let slice = unsafe { core::slice::from_raw_parts::(data_u8_ptr, data_size) }; + self.bytes.extend_from_slice(slice); + } + + /// Creates a pipeline state object from the stream. + /// + /// Safety: + /// - All unsafety invariants required by [`ID3D12Device2::CreatePipelineState`] must be upheld by the caller. + pub unsafe fn create_pipeline_state( + &mut self, + device: &ID3D12Device2, + ) -> windows::core::Result { + let stream_desc = D3D12_PIPELINE_STATE_STREAM_DESC { + SizeInBytes: self.bytes.len(), + pPipelineStateSubobjectStream: self.bytes.as_mut_ptr().cast(), + }; + + // Safety: lifetime on Self preserved the contents + // of the stream. Other unsafety invariants are upheld by the caller. + unsafe { device.CreatePipelineState(&stream_desc) } + } +} + +#[repr(C)] +#[derive(Debug)] +pub struct RenderPipelineStateStreamDesc<'a> { + pub root_signature: Option<&'a ID3D12RootSignature>, + pub pixel_shader: D3D12_SHADER_BYTECODE, + pub blend_state: D3D12_BLEND_DESC, + pub sample_mask: u32, + pub rasterizer_state: D3D12_RASTERIZER_DESC, + pub depth_stencil_state: D3D12_DEPTH_STENCIL_DESC, + pub primitive_topology_type: D3D12_PRIMITIVE_TOPOLOGY_TYPE, + pub rtv_formats: D3D12_RT_FORMAT_ARRAY, + pub dsv_format: DXGI_FORMAT, + pub sample_desc: DXGI_SAMPLE_DESC, + pub node_mask: u32, + pub cached_pso: D3D12_CACHED_PIPELINE_STATE, + pub flags: D3D12_PIPELINE_STATE_FLAGS, + pub view_instancing: Option, + + // Vertex pipeline specific + pub vertex_shader: D3D12_SHADER_BYTECODE, + pub input_layout: D3D12_INPUT_LAYOUT_DESC, + pub index_buffer_strip_cut_value: D3D12_INDEX_BUFFER_STRIP_CUT_VALUE, + pub stream_output: D3D12_STREAM_OUTPUT_DESC, + + // Mesh pipeline specific + pub task_shader: D3D12_SHADER_BYTECODE, + pub mesh_shader: D3D12_SHADER_BYTECODE, +} + +impl RenderPipelineStateStreamDesc<'_> { + pub fn to_stream(&self) -> RenderPipelineStateStream<'_> { + let mut stream = RenderPipelineStateStream::new(); + + // Importantly here, the ID3D12RootSignature _itself_ is the pointer we're + // trying to serialize into the stream, not a pointer to the pointer. + // + // This is correct because as_raw() returns turns that smart object into the raw + // pointer that _is_ the com object handle. + let root_sig_pointer = self + .root_signature + .map(|a| NonNull::new(a.as_raw()).unwrap()); + // Because the stream object borrows from self for its entire lifetime, + // it is safe to store the pointer into it. + stream.add_object(RootSignature(root_sig_pointer)); + + stream.add_object(self.blend_state); + stream.add_object(SampleMask(self.sample_mask)); + stream.add_object(self.rasterizer_state); + stream.add_object(self.depth_stencil_state); + stream.add_object(self.primitive_topology_type); + if self.rtv_formats.NumRenderTargets != 0 { + stream.add_object(self.rtv_formats); + } + if self.dsv_format != DXGI_FORMAT_UNKNOWN { + stream.add_object(self.dsv_format); + } + stream.add_object(self.sample_desc); + if self.node_mask != 0 { + stream.add_object(NodeMask(self.node_mask)); + } + if !self.cached_pso.pCachedBlob.is_null() { + stream.add_object(self.cached_pso); + } + stream.add_object(self.flags); + if let Some(view_instancing) = self.view_instancing { + stream.add_object(view_instancing); + } + if !self.pixel_shader.pShaderBytecode.is_null() { + stream.add_object(PixelShader(self.pixel_shader)); + } + if !self.vertex_shader.pShaderBytecode.is_null() { + stream.add_object(VertexShader(self.vertex_shader)); + stream.add_object(self.input_layout); + stream.add_object(self.index_buffer_strip_cut_value); + stream.add_object(self.stream_output); + } + if !self.task_shader.pShaderBytecode.is_null() { + stream.add_object(TaskShader(self.task_shader)); + } + if !self.mesh_shader.pShaderBytecode.is_null() { + stream.add_object(MeshShader(self.mesh_shader)); + } + + stream + } + + /// Returns a traditional D3D12_GRAPHICS_PIPELINE_STATE_DESC. + /// + /// Safety: + /// - This returned struct must not outlive self. + pub unsafe fn to_graphics_pipeline_descriptor(&self) -> D3D12_GRAPHICS_PIPELINE_STATE_DESC { + D3D12_GRAPHICS_PIPELINE_STATE_DESC { + pRootSignature: if let Some(rsig) = self.root_signature { + unsafe { borrow_interface_temporarily(rsig) } + } else { + ManuallyDrop::new(None) + }, + VS: self.vertex_shader, + PS: self.pixel_shader, + DS: D3D12_SHADER_BYTECODE::default(), + HS: D3D12_SHADER_BYTECODE::default(), + GS: D3D12_SHADER_BYTECODE::default(), + StreamOutput: self.stream_output, + BlendState: self.blend_state, + SampleMask: self.sample_mask, + RasterizerState: self.rasterizer_state, + DepthStencilState: self.depth_stencil_state, + InputLayout: self.input_layout, + IBStripCutValue: self.index_buffer_strip_cut_value, + PrimitiveTopologyType: self.primitive_topology_type, + NumRenderTargets: self.rtv_formats.NumRenderTargets, + RTVFormats: self.rtv_formats.RTFormats, + DSVFormat: self.dsv_format, + SampleDesc: self.sample_desc, + NodeMask: self.node_mask, + CachedPSO: self.cached_pso, + Flags: self.flags, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn wrappers() { + assert_eq!(size_of::(), size_of::()); + assert_eq!( + align_of::(), + align_of::() + ) + } + + implement_stream_object!(unsafe u16 => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE(1)); + implement_stream_object!(unsafe u32 => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE(2)); + implement_stream_object!(unsafe u64 => D3D12_PIPELINE_STATE_SUBOBJECT_TYPE(3)); + + #[test] + fn stream() { + let mut stream = RenderPipelineStateStream::new(); + + stream.add_object(42u16); + stream.add_object(84u32); + stream.add_object(168u64); + + assert_eq!(stream.bytes.len(), 32); + + // Object 1: u16 + + // Tag at the beginning + assert_eq!(&stream.bytes[0..4], &1u32.to_ne_bytes()); + // Data tucked in, aligned to the natural alignment of u16 + assert_eq!(&stream.bytes[4..6], &42u16.to_ne_bytes()); + // Padding to align the next subobject to an 8 byte boundary. + assert_eq!(&stream.bytes[6..8], &[0, 0]); + + // Object 2: u32 + + // Tag at the beginning + assert_eq!(&stream.bytes[8..12], &2u32.to_ne_bytes()); + // Data tucked in, aligned to the natural alignment of u32 + assert_eq!(&stream.bytes[12..16], &84u32.to_ne_bytes()); + + // Object 3: u64 + + // Tag at the beginning + assert_eq!(&stream.bytes[16..20], &3u32.to_ne_bytes()); + // Padding to align the u64 to an 8 byte boundary. + assert_eq!(&stream.bytes[20..24], &[0, 0, 0, 0]); + // Data tucked in, aligned to the natural alignment of u64 + assert_eq!(&stream.bytes[24..32], &168u64.to_ne_bytes()); + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/sampler.rs b/third_party/wgpu-hal-29.0.4/src/dx12/sampler.rs new file mode 100644 index 0000000..4b58a0b --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/sampler.rs @@ -0,0 +1,252 @@ +//! Sampler management for DX12. +//! +//! Nearly identical to the Vulkan sampler cache, with added descriptor heap management. + +use alloc::vec::Vec; + +use hashbrown::{hash_map::Entry, HashMap}; + +use ordered_float::OrderedFloat; +use parking_lot::Mutex; +use windows::Win32::Graphics::Direct3D12::*; + +use crate::dx12::HResult; + +/// The index of a sampler in the global sampler heap. +/// +/// This is a type-safe, transparent wrapper around a u32. +#[repr(transparent)] +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub(crate) struct SamplerIndex(u32); + +/// [`D3D12_SAMPLER_DESC`] is not hashable, so we wrap it in a newtype that is. +/// +/// We use [`OrderedFloat`] to allow for floating point values to be compared and +/// hashed in a defined way. +#[derive(Debug, Copy, Clone)] +struct HashableSamplerDesc(D3D12_SAMPLER_DESC); + +impl PartialEq for HashableSamplerDesc { + fn eq(&self, other: &Self) -> bool { + self.0.Filter == other.0.Filter + && self.0.AddressU == other.0.AddressU + && self.0.AddressV == other.0.AddressV + && self.0.AddressW == other.0.AddressW + && OrderedFloat(self.0.MipLODBias) == OrderedFloat(other.0.MipLODBias) + && self.0.MaxAnisotropy == other.0.MaxAnisotropy + && self.0.ComparisonFunc == other.0.ComparisonFunc + && self.0.BorderColor.map(OrderedFloat) == other.0.BorderColor.map(OrderedFloat) + && OrderedFloat(self.0.MinLOD) == OrderedFloat(other.0.MinLOD) + && OrderedFloat(self.0.MaxLOD) == OrderedFloat(other.0.MaxLOD) + } +} + +impl Eq for HashableSamplerDesc {} + +impl core::hash::Hash for HashableSamplerDesc { + fn hash(&self, state: &mut H) { + self.0.Filter.0.hash(state); + self.0.AddressU.0.hash(state); + self.0.AddressV.0.hash(state); + self.0.AddressW.0.hash(state); + OrderedFloat(self.0.MipLODBias).hash(state); + self.0.MaxAnisotropy.hash(state); + self.0.ComparisonFunc.0.hash(state); + self.0.BorderColor.map(OrderedFloat).hash(state); + OrderedFloat(self.0.MinLOD).hash(state); + OrderedFloat(self.0.MaxLOD).hash(state); + } +} + +/// Entry in the sampler cache. +struct CacheEntry { + index: SamplerIndex, + ref_count: u32, +} + +/// Container for the mutable management state of the sampler heap. +/// +/// We have this separated, using interior mutability, to allow for the outside world +/// to access the heap directly without needing to take the lock. +pub(crate) struct SamplerHeapState { + /// Mapping from the sampler description to the index within the heap and the refcount. + mapping: HashMap, + /// List of free sampler indices. + freelist: Vec, +} + +/// Global sampler heap for the device. +/// +/// As D3D12 only allows 2048 samplers to be in a single heap, we need to cache +/// samplers aggressively and refer to them in shaders by index. +pub(crate) struct SamplerHeap { + /// Mutable management state of the sampler heap. + state: Mutex, + + /// The heap itself. + heap: ID3D12DescriptorHeap, + /// The CPU-side handle to the first descriptor in the heap. + /// + /// Both the CPU and GPU handles point to the same descriptor, just in + /// different contexts. + heap_cpu_start_handle: D3D12_CPU_DESCRIPTOR_HANDLE, + /// The GPU-side handle to the first descriptor in the heap. + /// + /// Both the CPU and GPU handles point to the same descriptor, just in + /// different contexts. + heap_gpu_start_handle: D3D12_GPU_DESCRIPTOR_HANDLE, + + /// This is the device-specific size of sampler descriptors. + descriptor_stride: u32, +} + +impl SamplerHeap { + pub fn new( + device: &ID3D12Device, + private_caps: &super::PrivateCapabilities, + ) -> Result { + profiling::scope!("SamplerHeap::new"); + + // WARP can report this as 2M or more. We clamp it to 64k to be safe. + const SAMPLER_HEAP_SIZE_CLAMP: u32 = 64 * 1024; + + let max_unique_samplers = private_caps + .max_sampler_descriptor_heap_size + .min(SAMPLER_HEAP_SIZE_CLAMP); + + let desc = D3D12_DESCRIPTOR_HEAP_DESC { + Type: D3D12_DESCRIPTOR_HEAP_TYPE_SAMPLER, + NumDescriptors: max_unique_samplers, + Flags: D3D12_DESCRIPTOR_HEAP_FLAG_SHADER_VISIBLE, + NodeMask: 0, + }; + let heap = unsafe { device.CreateDescriptorHeap::(&desc) } + .into_device_result("Failed to create global GPU-Visible Sampler Descriptor Heap")?; + + let heap_cpu_start_handle = unsafe { heap.GetCPUDescriptorHandleForHeapStart() }; + let heap_gpu_start_handle = unsafe { heap.GetGPUDescriptorHandleForHeapStart() }; + + let descriptor_stride = + unsafe { device.GetDescriptorHandleIncrementSize(D3D12_DESCRIPTOR_HEAP_TYPE_SAMPLER) }; + + Ok(Self { + state: Mutex::new(SamplerHeapState { + mapping: HashMap::new(), + // Reverse so that samplers get allocated starting from zero. + freelist: (0..max_unique_samplers).map(SamplerIndex).rev().collect(), + }), + heap, + heap_cpu_start_handle, + heap_gpu_start_handle, + descriptor_stride, + }) + } + + /// Returns a reference to the raw descriptor heap. + pub fn heap(&self) -> &ID3D12DescriptorHeap { + &self.heap + } + + /// Returns a reference the handle to be bound to the descriptor table. + pub fn gpu_descriptor_table(&self) -> D3D12_GPU_DESCRIPTOR_HANDLE { + self.heap_gpu_start_handle + } + + /// Add a sampler with the given description to the heap. + /// + /// If the sampler already exists, the refcount is incremented and the existing index is returned. + /// + /// If the sampler does not exist, a new sampler is created and the index is returned. + /// + /// If the heap is full, an error is returned. + pub fn create_sampler( + &self, + device: &ID3D12Device, + desc: D3D12_SAMPLER_DESC, + ) -> Result { + profiling::scope!("SamplerHeap::create_sampler"); + + let hashable_desc = HashableSamplerDesc(desc); + + // Eagarly dereference the lock to allow split borrows. + let state = &mut *self.state.lock(); + + // Lookup the sampler in the mapping. + match state.mapping.entry(hashable_desc) { + Entry::Occupied(occupied_entry) => { + // We have found a match, so increment the refcount and return the index. + let entry = occupied_entry.into_mut(); + entry.ref_count += 1; + Ok(entry.index) + } + Entry::Vacant(vacant_entry) => { + // We need to create a new sampler. + + // Try to get a new index from the freelist. + let Some(index) = state.freelist.pop() else { + // If the freelist is empty, we have hit the maximum number of samplers. + log::error!("There is no more room in the global sampler heap for more unique samplers. Your device supports a maximum of {} unique samplers.", state.mapping.len()); + return Err(crate::DeviceError::OutOfMemory); + }; + + // Compute the CPU side handle for the new sampler. + let handle = D3D12_CPU_DESCRIPTOR_HANDLE { + ptr: self.heap_cpu_start_handle.ptr + + self.descriptor_stride as usize * index.0 as usize, + }; + + unsafe { + device.CreateSampler(&desc, handle); + } + + // Insert the new sampler into the mapping. + vacant_entry.insert(CacheEntry { + index, + ref_count: 1, + }); + + Ok(index) + } + } + } + + /// Decrement the refcount of the sampler with the given description. + /// + /// If the refcount reaches zero, the sampler is destroyed and the index is returned to the freelist. + /// + /// The provided index is checked against the index of the sampler with the given description, ensuring + /// that there isn't a clerical error from the caller. + pub fn destroy_sampler(&self, desc: D3D12_SAMPLER_DESC, provided_index: SamplerIndex) { + profiling::scope!("SamplerHeap::destroy_sampler"); + + // Eagarly dereference the lock to allow split borrows. + let state = &mut *self.state.lock(); + + // Get the index of the sampler to destroy. + let Entry::Occupied(mut hash_map_entry) = state.mapping.entry(HashableSamplerDesc(desc)) + else { + log::error!( + "Tried to destroy a sampler that doesn't exist. Sampler description: {desc:#?}" + ); + return; + }; + let cache_entry = hash_map_entry.get_mut(); + + // Ensure that the provided index matches the index of the sampler to destroy. + assert_eq!( + cache_entry.index, provided_index, + "Mismatched sampler index, this is an implementation bug" + ); + + // Decrement the refcount of the sampler. + cache_entry.ref_count -= 1; + + // If we are the last reference, remove the sampler from the mapping and return the index to the freelist. + // + // As samplers only exist as descriptors in the heap, there is nothing needed to be done to destroy the sampler. + if cache_entry.ref_count == 0 { + state.freelist.push(cache_entry.index); + hash_map_entry.remove(); + } + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/shader_compilation.rs b/third_party/wgpu-hal-29.0.4/src/dx12/shader_compilation.rs new file mode 100644 index 0000000..4b199b4 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/shader_compilation.rs @@ -0,0 +1,447 @@ +use alloc::{string::String, vec::Vec}; +use core::ffi::CStr; +use std::path::PathBuf; + +use crate::auxil::dxgi::result::HResult; +use thiserror::Error; +use windows::{ + core::{Interface, PCSTR, PCWSTR}, + Win32::Graphics::Direct3D::{Dxc, Fxc, ID3DBlob, D3D_SHADER_MACRO}, +}; + +pub(super) enum CompilerContainer { + Fxc(CompilerFxc), + DynamicDxc(CompilerDynamicDxc), + #[cfg_attr(not(static_dxc), allow(unused))] + StaticDxc(CompilerStaticDxc), +} + +pub(super) struct CompilerFxc { + fxc: FxcLib, +} + +pub(super) struct CompilerDynamicDxc { + max_shader_model: wgt::DxcShaderModel, + compiler: Dxc::IDxcCompiler3, + // Has to be held onto for the lifetime of the device otherwise shaders will fail to compile. + // Only needed when using dynamic linking. + _dxc: DxcLib, +} + +pub(super) struct CompilerStaticDxc { + max_shader_model: wgt::DxcShaderModel, + compiler: Dxc::IDxcCompiler3, +} + +#[derive(Debug, Error)] +pub(super) enum GetContainerError { + #[error(transparent)] + Device(#[from] crate::DeviceError), + #[error("Failed to load {0}: {1}")] + FailedToLoad(&'static str, libloading::Error), +} + +impl CompilerContainer { + pub(super) fn new_fxc() -> Result { + FxcLib::new_dynamic().map(|fxc| Self::Fxc(CompilerFxc { fxc })) + } + + pub(super) fn new_dynamic_dxc(dxc_path: PathBuf) -> Result { + let dxc = DxcLib::new_dynamic(dxc_path) + .map_err(|e| GetContainerError::FailedToLoad("dxcompiler.dll", e))?; + + let compiler = dxc.create_instance::()?; + + let (mut major, mut minor) = (1, 0); + // DXC 1.0 didn't support this. If the cast fails assume it is DXC 1.0. + if let Ok(version_info) = compiler.cast::() { + unsafe { + version_info.GetVersion(&mut major, &mut minor).unwrap(); + } + } + + Ok(Self::DynamicDxc(CompilerDynamicDxc { + max_shader_model: wgt::DxcShaderModel::from_dxc_version(major, minor), + compiler, + _dxc: dxc, + })) + } + + /// Creates a [`CompilerContainer`] that delegates to the statically-linked version of DXC. + pub(super) fn new_static_dxc() -> Result { + #[cfg(static_dxc)] + { + unsafe { + let compiler = dxc_create_instance::(|clsid, iid, ppv| { + windows_core::HRESULT(mach_dxcompiler_rs::DxcCreateInstance( + clsid.cast(), + iid.cast(), + ppv, + )) + })?; + + Ok(CompilerContainer::StaticDxc(CompilerStaticDxc { + max_shader_model: wgt::DxcShaderModel::V6_7, + compiler, + })) + } + } + #[cfg(not(static_dxc))] + { + panic!("Attempted to create a static DXC shader compiler, but the static-dxc feature was not enabled") + } + } + + pub(super) fn max_shader_model(&self) -> Option { + match self { + CompilerContainer::Fxc(..) => None, + CompilerContainer::DynamicDxc(CompilerDynamicDxc { + max_shader_model, .. + }) + | CompilerContainer::StaticDxc(CompilerStaticDxc { + max_shader_model, .. + }) => Some(max_shader_model.clone()), + } + } + + pub(super) fn compile( + &self, + device: &super::Device, + source: &str, + source_name: Option<&CStr>, + raw_ep: &str, + stage_bit: wgt::ShaderStages, + full_stage: &str, + ) -> Result { + match self { + CompilerContainer::Fxc(CompilerFxc { fxc }) => compile_fxc( + device, + source, + source_name, + raw_ep, + stage_bit, + full_stage, + fxc, + ), + CompilerContainer::DynamicDxc(CompilerDynamicDxc { compiler, .. }) + | CompilerContainer::StaticDxc(CompilerStaticDxc { compiler, .. }) => compile_dxc( + device, + source, + source_name, + raw_ep, + stage_bit, + full_stage, + compiler, + ), + } + } +} + +type D3DCompileFn = unsafe extern "system" fn( + psrcdata: *const core::ffi::c_void, + srcdatasize: usize, + psourcename: PCSTR, + pdefines: *const D3D_SHADER_MACRO, + pinclude: *mut core::ffi::c_void, + pentrypoint: PCSTR, + ptarget: PCSTR, + flags1: u32, + flags2: u32, + ppcode: *mut *mut core::ffi::c_void, + pperrormsgs: *mut *mut core::ffi::c_void, +) -> windows_core::HRESULT; + +#[derive(Debug)] +struct FxcLib { + // `d3dcompile_fn` points into `_lib`, so `_lib` must be held for as long + // as we want to keep compiling shaders with FXC. + _lib: crate::dx12::DynLib, + d3dcompile_fn: D3DCompileFn, +} + +impl FxcLib { + const PATH: &str = "d3dcompiler_47.dll"; + + fn new_dynamic() -> Result { + unsafe { + let lib = crate::dx12::DynLib::new(Self::PATH) + .map_err(|e| GetContainerError::FailedToLoad(FxcLib::PATH, e))?; + let d3dcompile_fn: D3DCompileFn = *lib.get::(c"D3DCompile".to_bytes())?; + + Ok(Self { + _lib: lib, + d3dcompile_fn, + }) + } + } + + #[allow(clippy::too_many_arguments)] + fn compile( + &self, + source: &str, + source_name: Option<&CStr>, + raw_ep: &str, + full_stage: &str, + compile_flags: u32, + shader_data: &mut Option, + error: &mut Option, + ) -> Result, crate::DeviceError> { + unsafe { + let raw_ep = alloc::ffi::CString::new(raw_ep).unwrap(); + let full_stage = alloc::ffi::CString::new(full_stage).unwrap(); + + // If no name has been set, D3DCompile wants the null pointer. + let source_name = source_name + .map(|cstr| cstr.as_ptr().cast()) + .unwrap_or(core::ptr::null()); + + let shader_data: *mut Option = shader_data; + let error: *mut Option = error; + + { + profiling::scope!("Fxc::D3DCompile"); + Ok((self.d3dcompile_fn)( + source.as_ptr().cast(), + source.len(), + PCSTR(source_name), + core::ptr::null(), + core::ptr::null_mut(), + PCSTR(raw_ep.as_ptr().cast()), + PCSTR(full_stage.as_ptr().cast()), + compile_flags, + 0, + shader_data.cast(), + error.cast(), + ) + .ok()) + } + } + } +} + +fn compile_fxc( + device: &super::Device, + source: &str, + source_name: Option<&CStr>, + raw_ep: &str, + stage_bit: wgt::ShaderStages, + full_stage: &str, + fxc: &FxcLib, +) -> Result { + profiling::scope!("compile_fxc"); + let mut compile_flags = Fxc::D3DCOMPILE_ENABLE_STRICTNESS; + if device + .shared + .private_caps + .instance_flags + .contains(wgt::InstanceFlags::DEBUG) + { + compile_flags |= Fxc::D3DCOMPILE_DEBUG | Fxc::D3DCOMPILE_SKIP_OPTIMIZATION; + } + + let mut shader_data = None; + let mut error = None; + let hr = fxc.compile( + source, + source_name, + raw_ep, + full_stage, + compile_flags, + &mut shader_data, + &mut error, + )?; + + match hr { + Ok(()) => { + let shader_data = shader_data.unwrap(); + Ok(super::CompiledShader::Fxc(shader_data)) + } + Err(e) => { + let mut full_msg = format!("FXC D3DCompile error ({e})"); + if let Some(error) = error { + use core::fmt::Write as _; + let message = unsafe { + core::slice::from_raw_parts( + error.GetBufferPointer().cast(), + error.GetBufferSize(), + ) + }; + let _ = write!(full_msg, ": {}", String::from_utf8_lossy(message)); + } + Err(crate::PipelineError::Linkage(stage_bit, full_msg)) + } + } +} + +trait DxcObj: Interface { + const CLSID: windows::core::GUID; +} +impl DxcObj for Dxc::IDxcCompiler3 { + const CLSID: windows::core::GUID = Dxc::CLSID_DxcCompiler; +} +impl DxcObj for Dxc::IDxcUtils { + const CLSID: windows::core::GUID = Dxc::CLSID_DxcUtils; +} +impl DxcObj for Dxc::IDxcValidator { + const CLSID: windows::core::GUID = Dxc::CLSID_DxcValidator; +} + +#[derive(Debug)] +struct DxcLib { + lib: crate::dx12::DynLib, +} + +impl DxcLib { + fn new_dynamic(lib_path: PathBuf) -> Result { + unsafe { crate::dx12::DynLib::new(lib_path).map(|lib| Self { lib }) } + } + + pub fn create_instance(&self) -> Result { + unsafe { + type DxcCreateInstanceFn = unsafe extern "system" fn( + rclsid: *const windows_core::GUID, + riid: *const windows_core::GUID, + ppv: *mut *mut core::ffi::c_void, + ) + -> windows_core::HRESULT; + + let func: libloading::Symbol = + self.lib.get(c"DxcCreateInstance".to_bytes())?; + dxc_create_instance::(|clsid, iid, ppv| func(clsid, iid, ppv)) + } + } +} + +/// Invokes the provided library function to create a DXC object. +unsafe fn dxc_create_instance( + f: impl Fn( + *const windows_core::GUID, + *const windows_core::GUID, + *mut *mut core::ffi::c_void, + ) -> windows_core::HRESULT, +) -> Result { + let mut result__ = None; + f(&T::CLSID, &T::IID, <*mut _>::cast(&mut result__)) + .ok() + .into_device_result("DxcCreateInstance")?; + result__.ok_or(crate::DeviceError::Unexpected) +} + +/// Owned PCWSTR +#[allow(clippy::upper_case_acronyms)] +struct OPCWSTR { + inner: Vec, +} + +impl OPCWSTR { + fn new(s: &str) -> Self { + let mut inner: Vec<_> = s.encode_utf16().collect(); + inner.push(0); + Self { inner } + } + + fn ptr(&self) -> PCWSTR { + PCWSTR(self.inner.as_ptr()) + } +} + +fn get_output( + res: &Dxc::IDxcResult, + kind: Dxc::DXC_OUT_KIND, +) -> Result { + let mut result__: Option = None; + unsafe { res.GetOutput::(kind, &mut None, <*mut _>::cast(&mut result__)) } + .into_device_result("GetOutput")?; + result__.ok_or(crate::DeviceError::Unexpected) +} + +fn as_err_str(blob: &Dxc::IDxcBlobUtf8) -> Result<&str, crate::DeviceError> { + let ptr = unsafe { blob.GetStringPointer() }; + let len = unsafe { blob.GetStringLength() }; + core::str::from_utf8(unsafe { core::slice::from_raw_parts(ptr.0, len) }) + .map_err(|_| crate::DeviceError::Unexpected) +} + +fn compile_dxc( + device: &crate::dx12::Device, + source: &str, + source_name: Option<&CStr>, + raw_ep: &str, + stage_bit: wgt::ShaderStages, + full_stage: &str, + compiler: &Dxc::IDxcCompiler3, +) -> Result { + profiling::scope!("compile_dxc"); + + let source_name = source_name.and_then(|cstr| cstr.to_str().ok()); + + let source_name = source_name.map(OPCWSTR::new); + let raw_ep = OPCWSTR::new(raw_ep); + let full_stage = OPCWSTR::new(full_stage); + + let mut compile_args = arrayvec::ArrayVec::::new_const(); + + if let Some(source_name) = source_name.as_ref() { + compile_args.push(source_name.ptr()) + } + + compile_args.extend([ + windows::core::w!("-E"), + raw_ep.ptr(), + windows::core::w!("-T"), + full_stage.ptr(), + windows::core::w!("-HV"), + windows::core::w!("2018"), // Use HLSL 2018, Naga doesn't supported 2021 yet. + windows::core::w!("-no-warnings"), + Dxc::DXC_ARG_ENABLE_STRICTNESS, + ]); + + if device + .shared + .private_caps + .instance_flags + .contains(wgt::InstanceFlags::DEBUG) + && !device + .shared + .private_caps + .workarounds + .avoid_shader_debug_info + { + compile_args.push(Dxc::DXC_ARG_DEBUG); + compile_args.push(Dxc::DXC_ARG_SKIP_OPTIMIZATIONS); + } + + if device.features.contains(wgt::Features::SHADER_F16) { + compile_args.push(windows::core::w!("-enable-16bit-types")); + } + + let buffer = Dxc::DxcBuffer { + Ptr: source.as_ptr().cast(), + Size: source.len(), + Encoding: Dxc::DXC_CP_UTF8.0, + }; + + let compile_res: Dxc::IDxcResult = + unsafe { compiler.Compile(&buffer, Some(&compile_args), None) } + .into_device_result("Compile")?; + + drop(compile_args); + drop(source_name); + drop(raw_ep); + drop(full_stage); + + let err_blob = get_output::(&compile_res, Dxc::DXC_OUT_ERRORS)?; + + let len = unsafe { err_blob.GetStringLength() }; + if len != 0 { + let err = as_err_str(&err_blob)?; + return Err(crate::PipelineError::Linkage( + stage_bit, + format!("DXC compile error: {err}"), + )); + } + + let blob = get_output::(&compile_res, Dxc::DXC_OUT_OBJECT)?; + + Ok(crate::dx12::CompiledShader::Dxc(blob)) +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/suballocation.rs b/third_party/wgpu-hal-29.0.4/src/dx12/suballocation.rs new file mode 100644 index 0000000..89c7dc4 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/suballocation.rs @@ -0,0 +1,603 @@ +use alloc::sync::Arc; + +use gpu_allocator::{d3d12::AllocationCreateDesc, MemoryLocation}; +use parking_lot::Mutex; +use windows::Win32::Graphics::{Direct3D12, Dxgi}; + +use crate::{ + auxil::dxgi::{name::ObjectExt as _, result::HResult as _}, + dx12::conv, + AllocationSizes, +}; + +#[derive(Debug)] +pub(crate) enum AllocationType { + Buffer, + Texture, + AccelerationStructure, +} + +#[derive(Debug)] +enum AllocationInner { + /// This resource is suballocated from a heap. + Placed { + inner: gpu_allocator::d3d12::Allocation, + }, + /// This resource is a committed resource and does not belong to a + /// suballocated heap. We store an approximate size, so we can manage our counters + /// correctly. + /// + /// This is only used for Intel Xe drivers, which have a bug that + /// prevents suballocation from working correctly. + Committed { size: u64 }, +} + +#[derive(Debug)] +pub(crate) struct Allocation { + inner: AllocationInner, + ty: AllocationType, +} + +impl Allocation { + pub fn placed(inner: gpu_allocator::d3d12::Allocation, ty: AllocationType) -> Self { + Self { + inner: AllocationInner::Placed { inner }, + ty, + } + } + + pub fn none(ty: AllocationType, size: u64) -> Self { + Self { + inner: AllocationInner::Committed { size }, + ty, + } + } + + pub fn size(&self) -> u64 { + match self.inner { + AllocationInner::Placed { ref inner } => inner.size(), + AllocationInner::Committed { size } => size, + } + } +} + +#[derive(Clone)] +pub(crate) struct Allocator { + inner: Arc>, + device_memblock_size: u64, + host_memblock_size: u64, + pub memory_budget_thresholds: wgt::MemoryBudgetThresholds, +} + +impl Allocator { + pub(crate) fn new( + raw: &Direct3D12::ID3D12Device, + memory_hints: &wgt::MemoryHints, + memory_budget_thresholds: wgt::MemoryBudgetThresholds, + ) -> Result { + let allocation_sizes = AllocationSizes::from_memory_hints(memory_hints); + let device_memblock_size = allocation_sizes.min_device_memblock_size; + let host_memblock_size = allocation_sizes.min_host_memblock_size; + + let allocator_desc = gpu_allocator::d3d12::AllocatorCreateDesc { + device: gpu_allocator::d3d12::ID3D12DeviceVersion::Device(raw.clone()), + debug_settings: Default::default(), + allocation_sizes: allocation_sizes.into(), + }; + + let allocator = gpu_allocator::d3d12::Allocator::new(&allocator_desc).inspect_err(|e| { + log::error!("Failed to create d3d12 allocator, error: {e}"); + })?; + + Ok(Self { + inner: Arc::new(Mutex::new(allocator)), + device_memblock_size, + host_memblock_size, + memory_budget_thresholds, + }) + } + + pub(crate) fn generate_report(&self) -> wgt::AllocatorReport { + let mut upstream = self.inner.lock().generate_report(); + + let allocations = upstream + .allocations + .iter_mut() + .map(|alloc| wgt::AllocationReport { + name: core::mem::take(&mut alloc.name), + offset: alloc.offset, + size: alloc.size, + }) + .collect(); + + let blocks = upstream + .blocks + .iter() + .map(|block| wgt::MemoryBlockReport { + size: block.size, + allocations: block.allocations.clone(), + }) + .collect(); + + wgt::AllocatorReport { + allocations, + blocks, + total_allocated_bytes: upstream.total_allocated_bytes, + total_reserved_bytes: upstream.total_capacity_bytes, + } + } +} + +/// To allow us to construct buffers from both a `Device` and `CommandEncoder` +/// without needing each function to take a million arguments, we create a +/// borrowed context struct that contains the relevant members. +pub(crate) struct DeviceAllocationContext<'a> { + pub(crate) raw: &'a Direct3D12::ID3D12Device, + pub(crate) shared: &'a super::DeviceShared, + pub(crate) mem_allocator: &'a Allocator, + pub(crate) counters: &'a wgt::HalCounters, +} + +impl<'a> From<&'a super::Device> for DeviceAllocationContext<'a> { + fn from(device: &'a super::Device) -> Self { + Self { + raw: &device.raw, + shared: &device.shared, + mem_allocator: &device.mem_allocator, + counters: &device.counters, + } + } +} + +impl<'a> From<&'a super::CommandEncoder> for DeviceAllocationContext<'a> { + fn from(encoder: &'a super::CommandEncoder) -> Self { + Self { + raw: &encoder.device, + shared: &encoder.shared, + mem_allocator: &encoder.mem_allocator, + counters: &encoder.counters, + } + } +} + +impl<'a> DeviceAllocationContext<'a> { + /////////////////////// + // Resource Creation // + /////////////////////// + + pub(crate) fn create_buffer( + &self, + desc: &crate::BufferDescriptor, + ) -> Result<(Direct3D12::ID3D12Resource, Allocation), crate::DeviceError> { + let is_cpu_read = desc.usage.contains(wgt::BufferUses::MAP_READ); + let is_cpu_write = desc.usage.contains(wgt::BufferUses::MAP_WRITE); + + let location = match (is_cpu_read, is_cpu_write) { + (true, true) => MemoryLocation::CpuToGpu, + (true, false) => MemoryLocation::GpuToCpu, + (false, true) => MemoryLocation::CpuToGpu, + (false, false) => MemoryLocation::GpuOnly, + }; + + let raw_desc = conv::map_buffer_descriptor(desc); + let allocation_info = + self.error_if_would_oom_on_resource_allocation(&raw_desc, location)?; + + let (resource, allocation) = if self.shared.private_caps.suballocation_supported { + self.create_placed_buffer(desc, raw_desc, allocation_info, location)? + } else { + self.create_committed_buffer(raw_desc, location)? + }; + + if let Some(label) = desc.label { + resource.set_name(label)?; + } + + self.counters.buffer_memory.add(allocation.size() as isize); + + Ok((resource, allocation)) + } + + pub(crate) fn create_texture( + &self, + desc: &crate::TextureDescriptor, + raw_desc: Direct3D12::D3D12_RESOURCE_DESC, + ) -> Result<(Direct3D12::ID3D12Resource, Allocation), crate::DeviceError> { + let location = MemoryLocation::GpuOnly; + let allocation_info = + self.error_if_would_oom_on_resource_allocation(&raw_desc, location)?; + + let (resource, allocation) = if self.shared.private_caps.suballocation_supported { + self.create_placed_texture(desc, raw_desc, allocation_info, location)? + } else { + self.create_committed_texture(desc, raw_desc)? + }; + + if let Some(label) = desc.label { + resource.set_name(label)?; + } + + self.counters.texture_memory.add(allocation.size() as isize); + + Ok((resource, allocation)) + } + + pub(crate) fn create_acceleration_structure( + &self, + desc: &crate::AccelerationStructureDescriptor, + raw_desc: Direct3D12::D3D12_RESOURCE_DESC, + ) -> Result<(Direct3D12::ID3D12Resource, Allocation), crate::DeviceError> { + let location = MemoryLocation::GpuOnly; + let allocation_info = + self.error_if_would_oom_on_resource_allocation(&raw_desc, location)?; + + let (resource, allocation) = if self.shared.private_caps.suballocation_supported { + self.create_placed_acceleration_structure(desc, raw_desc, allocation_info, location)? + } else { + self.create_committed_acceleration_structure(desc, raw_desc)? + }; + + if let Some(label) = desc.label { + resource.set_name(label)?; + } + + self.counters + .acceleration_structure_memory + .add(allocation.size() as isize); + + Ok((resource, allocation)) + } + + ////////////////////////// + // Resource Destruction // + ////////////////////////// + + pub(crate) fn free_resource( + &self, + resource: Direct3D12::ID3D12Resource, + allocation: Allocation, + ) { + // Make sure the resource is released before we free the allocation. + drop(resource); + + let counter = match allocation.ty { + AllocationType::Buffer => &self.counters.buffer_memory, + AllocationType::Texture => &self.counters.texture_memory, + AllocationType::AccelerationStructure => &self.counters.acceleration_structure_memory, + }; + counter.sub(allocation.size() as isize); + + if let AllocationInner::Placed { inner } = allocation.inner { + match self.mem_allocator.inner.lock().free(inner) { + Ok(_) => (), + // TODO: Don't panic here + Err(e) => panic!("Failed to destroy dx12 {:?}, {e}", allocation.ty), + }; + } + } + + /////////////////////////////// + // Placed Resource Creation /// + /////////////////////////////// + + fn create_placed_buffer( + &self, + desc: &crate::BufferDescriptor<'_>, + raw_desc: Direct3D12::D3D12_RESOURCE_DESC, + allocation_info: Direct3D12::D3D12_RESOURCE_ALLOCATION_INFO, + location: MemoryLocation, + ) -> Result<(Direct3D12::ID3D12Resource, Allocation), crate::DeviceError> { + let name = desc.label.unwrap_or("Unlabeled buffer"); + + let mut allocator = self.mem_allocator.inner.lock(); + + let allocation_desc = AllocationCreateDesc { + name, + location, + size: allocation_info.SizeInBytes, + alignment: allocation_info.Alignment, + resource_category: gpu_allocator::d3d12::ResourceCategory::from(&raw_desc), + }; + + let allocation = allocator.allocate(&allocation_desc)?; + let mut resource = None; + unsafe { + self.raw.CreatePlacedResource( + allocation.heap(), + allocation.offset(), + &raw_desc, + Direct3D12::D3D12_RESOURCE_STATE_COMMON, + None, + &mut resource, + ) + } + .into_device_result("Placed buffer creation")?; + + let resource = resource.ok_or(crate::DeviceError::Unexpected)?; + let wrapped_allocation = Allocation::placed(allocation, AllocationType::Buffer); + + Ok((resource, wrapped_allocation)) + } + + fn create_placed_texture( + &self, + desc: &crate::TextureDescriptor<'_>, + raw_desc: Direct3D12::D3D12_RESOURCE_DESC, + allocation_info: Direct3D12::D3D12_RESOURCE_ALLOCATION_INFO, + location: MemoryLocation, + ) -> Result<(Direct3D12::ID3D12Resource, Allocation), crate::DeviceError> { + let name = desc.label.unwrap_or("Unlabeled texture"); + + let mut allocator = self.mem_allocator.inner.lock(); + + let allocation_desc = AllocationCreateDesc { + name, + location, + size: allocation_info.SizeInBytes, + alignment: allocation_info.Alignment, + resource_category: gpu_allocator::d3d12::ResourceCategory::from(&raw_desc), + }; + + let allocation = allocator.allocate(&allocation_desc)?; + let mut resource = None; + unsafe { + self.raw.CreatePlacedResource( + allocation.heap(), + allocation.offset(), + &raw_desc, + Direct3D12::D3D12_RESOURCE_STATE_COMMON, + None, // clear value + &mut resource, + ) + } + .into_device_result("Placed texture creation")?; + + let resource = resource.ok_or(crate::DeviceError::Unexpected)?; + let wrapped_allocation = Allocation::placed(allocation, AllocationType::Texture); + + Ok((resource, wrapped_allocation)) + } + + fn create_placed_acceleration_structure( + &self, + desc: &crate::AccelerationStructureDescriptor<'_>, + raw_desc: Direct3D12::D3D12_RESOURCE_DESC, + allocation_info: Direct3D12::D3D12_RESOURCE_ALLOCATION_INFO, + location: MemoryLocation, + ) -> Result<(Direct3D12::ID3D12Resource, Allocation), crate::DeviceError> { + let name = desc.label.unwrap_or("Unlabeled acceleration structure"); + + let mut allocator = self.mem_allocator.inner.lock(); + + let allocation_desc = AllocationCreateDesc { + name, + location, + size: allocation_info.SizeInBytes, + alignment: allocation_info.Alignment, + resource_category: gpu_allocator::d3d12::ResourceCategory::from(&raw_desc), + }; + + let allocation = allocator.allocate(&allocation_desc)?; + let mut resource = None; + unsafe { + self.raw.CreatePlacedResource( + allocation.heap(), + allocation.offset(), + &raw_desc, + Direct3D12::D3D12_RESOURCE_STATE_RAYTRACING_ACCELERATION_STRUCTURE, + None, + &mut resource, + ) + } + .into_device_result("Placed acceleration structure creation")?; + + let resource = resource.ok_or(crate::DeviceError::Unexpected)?; + let wrapped_allocation = + Allocation::placed(allocation, AllocationType::AccelerationStructure); + + Ok((resource, wrapped_allocation)) + } + + ///////////////////////////////// + // Committed Resource Creation // + ///////////////////////////////// + + fn create_committed_buffer( + &self, + raw_desc: Direct3D12::D3D12_RESOURCE_DESC, + location: MemoryLocation, + ) -> Result<(Direct3D12::ID3D12Resource, Allocation), crate::DeviceError> { + let is_uma = matches!( + self.shared.private_caps.memory_architecture, + crate::dx12::MemoryArchitecture::Unified { .. } + ); + + let heap_properties = Direct3D12::D3D12_HEAP_PROPERTIES { + Type: Direct3D12::D3D12_HEAP_TYPE_CUSTOM, + CPUPageProperty: match location { + MemoryLocation::GpuOnly => Direct3D12::D3D12_CPU_PAGE_PROPERTY_NOT_AVAILABLE, + MemoryLocation::CpuToGpu => Direct3D12::D3D12_CPU_PAGE_PROPERTY_WRITE_COMBINE, + MemoryLocation::GpuToCpu => Direct3D12::D3D12_CPU_PAGE_PROPERTY_WRITE_BACK, + _ => unreachable!(), + }, + MemoryPoolPreference: match (is_uma, location) { + // On dedicated GPUs, we only use L1 for GPU-only allocations. + (false, MemoryLocation::GpuOnly) => Direct3D12::D3D12_MEMORY_POOL_L1, + (_, _) => Direct3D12::D3D12_MEMORY_POOL_L0, + }, + CreationNodeMask: 0, + VisibleNodeMask: 0, + }; + + let mut resource = None; + + unsafe { + self.raw.CreateCommittedResource( + &heap_properties, + if self.shared.private_caps.heap_create_not_zeroed { + Direct3D12::D3D12_HEAP_FLAG_CREATE_NOT_ZEROED + } else { + Direct3D12::D3D12_HEAP_FLAG_NONE + }, + &raw_desc, + Direct3D12::D3D12_RESOURCE_STATE_COMMON, + None, + &mut resource, + ) + } + .into_device_result("Committed buffer creation")?; + + let resource = resource.ok_or(crate::DeviceError::Unexpected)?; + let wrapped_allocation = Allocation::none(AllocationType::Buffer, raw_desc.Width); + + Ok((resource, wrapped_allocation)) + } + + fn create_committed_texture( + &self, + desc: &crate::TextureDescriptor, + raw_desc: Direct3D12::D3D12_RESOURCE_DESC, + ) -> Result<(Direct3D12::ID3D12Resource, Allocation), crate::DeviceError> { + let heap_properties = Direct3D12::D3D12_HEAP_PROPERTIES { + Type: Direct3D12::D3D12_HEAP_TYPE_CUSTOM, + CPUPageProperty: Direct3D12::D3D12_CPU_PAGE_PROPERTY_NOT_AVAILABLE, + MemoryPoolPreference: match self.shared.private_caps.memory_architecture { + crate::dx12::MemoryArchitecture::NonUnified => Direct3D12::D3D12_MEMORY_POOL_L1, + crate::dx12::MemoryArchitecture::Unified { .. } => Direct3D12::D3D12_MEMORY_POOL_L0, + }, + CreationNodeMask: 0, + VisibleNodeMask: 0, + }; + + let mut resource = None; + + unsafe { + self.raw.CreateCommittedResource( + &heap_properties, + if self.shared.private_caps.heap_create_not_zeroed { + Direct3D12::D3D12_HEAP_FLAG_CREATE_NOT_ZEROED + } else { + Direct3D12::D3D12_HEAP_FLAG_NONE + }, + &raw_desc, + Direct3D12::D3D12_RESOURCE_STATE_COMMON, + None, // clear value + &mut resource, + ) + } + .into_device_result("Committed texture creation")?; + + let resource = resource.ok_or(crate::DeviceError::Unexpected)?; + let wrapped_allocation = Allocation::none( + AllocationType::Texture, + desc.format.theoretical_memory_footprint(desc.size), + ); + + Ok((resource, wrapped_allocation)) + } + + fn create_committed_acceleration_structure( + &self, + desc: &crate::AccelerationStructureDescriptor, + raw_desc: Direct3D12::D3D12_RESOURCE_DESC, + ) -> Result<(Direct3D12::ID3D12Resource, Allocation), crate::DeviceError> { + let heap_properties = Direct3D12::D3D12_HEAP_PROPERTIES { + Type: Direct3D12::D3D12_HEAP_TYPE_CUSTOM, + CPUPageProperty: Direct3D12::D3D12_CPU_PAGE_PROPERTY_NOT_AVAILABLE, + MemoryPoolPreference: match self.shared.private_caps.memory_architecture { + crate::dx12::MemoryArchitecture::NonUnified => Direct3D12::D3D12_MEMORY_POOL_L1, + crate::dx12::MemoryArchitecture::Unified { .. } => Direct3D12::D3D12_MEMORY_POOL_L0, + }, + CreationNodeMask: 0, + VisibleNodeMask: 0, + }; + + let mut resource = None; + + unsafe { + self.raw.CreateCommittedResource( + &heap_properties, + if self.shared.private_caps.heap_create_not_zeroed { + Direct3D12::D3D12_HEAP_FLAG_CREATE_NOT_ZEROED + } else { + Direct3D12::D3D12_HEAP_FLAG_NONE + }, + &raw_desc, + Direct3D12::D3D12_RESOURCE_STATE_RAYTRACING_ACCELERATION_STRUCTURE, + None, + &mut resource, + ) + } + .into_device_result("Committed acceleration structure creation")?; + + let resource = resource.ok_or(crate::DeviceError::Unexpected)?; + let wrapped_allocation = Allocation::none(AllocationType::AccelerationStructure, desc.size); + + Ok((resource, wrapped_allocation)) + } + + fn error_if_would_oom_on_resource_allocation( + &self, + desc: &Direct3D12::D3D12_RESOURCE_DESC, + location: MemoryLocation, + ) -> Result { + let allocation_info = unsafe { + self.raw + .GetResourceAllocationInfo(0, core::slice::from_ref(desc)) + }; + + // Some versions of WARP return SizeInBytes == 0 for very large + // allocations. Proceeding to attempt to allocate a zero-sized resource + // will result in a device lost error, so it seems preferable to return + // an out of memory error now. + if allocation_info.SizeInBytes == 0 { + return Err(crate::DeviceError::OutOfMemory); + } + + let Some(threshold) = self + .mem_allocator + .memory_budget_thresholds + .for_resource_creation + else { + return Ok(allocation_info); + }; + + let memory_segment_group = match location { + MemoryLocation::Unknown => unreachable!(), + MemoryLocation::GpuOnly => Dxgi::DXGI_MEMORY_SEGMENT_GROUP_LOCAL, + MemoryLocation::CpuToGpu | MemoryLocation::GpuToCpu => { + match self.shared.private_caps.memory_architecture { + super::MemoryArchitecture::Unified { .. } => { + Dxgi::DXGI_MEMORY_SEGMENT_GROUP_LOCAL + } + super::MemoryArchitecture::NonUnified => { + Dxgi::DXGI_MEMORY_SEGMENT_GROUP_NON_LOCAL + } + } + } + }; + + let info = self + .shared + .adapter + .query_video_memory_info(memory_segment_group)?; + + let memblock_size = match location { + MemoryLocation::Unknown => unreachable!(), + MemoryLocation::GpuOnly => self.mem_allocator.device_memblock_size, + MemoryLocation::CpuToGpu | MemoryLocation::GpuToCpu => { + self.mem_allocator.host_memblock_size + } + }; + + if info + .CurrentUsage + .checked_add(allocation_info.SizeInBytes.max(memblock_size)) + .is_none_or(|usage| usage >= info.Budget / 100 * threshold as u64) + { + return Err(crate::DeviceError::OutOfMemory); + } + + Ok(allocation_info) + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/types.rs b/third_party/wgpu-hal-29.0.4/src/dx12/types.rs new file mode 100644 index 0000000..5270c6c --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/types.rs @@ -0,0 +1,39 @@ +#![allow(non_camel_case_types)] +#![allow(non_snake_case)] + +use windows::Win32::Graphics::Dxgi; + +windows_core::imp::define_interface!( + ISwapChainPanelNative, + ISwapChainPanelNative_Vtbl, + 0x63aad0b8_7c24_40ff_85a8_640d944cc325 +); +impl core::ops::Deref for ISwapChainPanelNative { + type Target = windows_core::IUnknown; + fn deref(&self) -> &Self::Target { + unsafe { core::mem::transmute(self) } + } +} +windows_core::imp::interface_hierarchy!(ISwapChainPanelNative, windows_core::IUnknown); +impl ISwapChainPanelNative { + pub unsafe fn SetSwapChain(&self, swap_chain: P0) -> windows_core::Result<()> + where + P0: windows_core::Param, + { + unsafe { + (windows_core::Interface::vtable(self).SetSwapChain)( + windows_core::Interface::as_raw(self), + swap_chain.param().abi(), + ) + } + .ok() + } +} +#[repr(C)] +pub struct ISwapChainPanelNative_Vtbl { + pub base__: windows_core::IUnknown_Vtbl, + pub SetSwapChain: unsafe extern "system" fn( + swap_chain_panel_native: *mut core::ffi::c_void, + swap_chain: *mut core::ffi::c_void, + ) -> windows_core::HRESULT, +} diff --git a/third_party/wgpu-hal-29.0.4/src/dx12/view.rs b/third_party/wgpu-hal-29.0.4/src/dx12/view.rs new file mode 100644 index 0000000..e541ac2 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dx12/view.rs @@ -0,0 +1,340 @@ +use windows::Win32::Graphics::{Direct3D12, Dxgi}; + +use crate::auxil; + +pub(super) struct ViewDescriptor { + dimension: wgt::TextureViewDimension, + pub aspects: crate::FormatAspects, + pub rtv_dsv_format: Dxgi::Common::DXGI_FORMAT, + srv_uav_format: Option, + multisampled: bool, + array_layer_base: u32, + array_layer_count: u32, + mip_level_base: u32, + mip_level_count: u32, +} + +impl crate::TextureViewDescriptor<'_> { + pub(super) fn to_internal(&self, texture: &super::Texture) -> ViewDescriptor { + let aspects = crate::FormatAspects::new(texture.format, self.range.aspect); + + ViewDescriptor { + dimension: self.dimension, + aspects, + rtv_dsv_format: auxil::dxgi::conv::map_texture_format(self.format), + srv_uav_format: auxil::dxgi::conv::map_texture_format_for_srv_uav(self.format, aspects), + multisampled: texture.sample_count > 1, + mip_level_base: self.range.base_mip_level, + mip_level_count: self.range.mip_level_count.unwrap_or(!0), + array_layer_base: self.range.base_array_layer, + array_layer_count: self.range.array_layer_count.unwrap_or(!0), + } + } +} + +fn aspects_to_plane(aspects: crate::FormatAspects) -> u32 { + match aspects { + crate::FormatAspects::STENCIL => 1, + crate::FormatAspects::PLANE_1 => 1, + crate::FormatAspects::PLANE_2 => 2, + _ => 0, + } +} + +impl ViewDescriptor { + pub(crate) unsafe fn to_srv(&self) -> Option { + let mut desc = Direct3D12::D3D12_SHADER_RESOURCE_VIEW_DESC { + Format: self.srv_uav_format?, + ViewDimension: Direct3D12::D3D12_SRV_DIMENSION_UNKNOWN, + Shader4ComponentMapping: Direct3D12::D3D12_DEFAULT_SHADER_4_COMPONENT_MAPPING, + Anonymous: Default::default(), + }; + + match self.dimension { + wgt::TextureViewDimension::D1 => { + desc.ViewDimension = Direct3D12::D3D12_SRV_DIMENSION_TEXTURE1D; + desc.Anonymous.Texture1D = Direct3D12::D3D12_TEX1D_SRV { + MostDetailedMip: self.mip_level_base, + MipLevels: self.mip_level_count, + ResourceMinLODClamp: 0.0, + } + } + /* + wgt::TextureViewDimension::D1Array => { + desc.ViewDimension = Direct3D12::D3D12_SRV_DIMENSION_TEXTURE1DARRAY; + desc.Anonymous.Texture1DArray = Direct3D12::D3D12_TEX1D_ARRAY_SRV { + MostDetailedMip: self.mip_level_base, + MipLevels: self.mip_level_count, + FirstArraySlice: self.array_layer_base, + ArraySize: self.array_layer_count, + ResourceMinLODClamp: 0.0, + } + } + */ + wgt::TextureViewDimension::D2 if self.multisampled && self.array_layer_base == 0 => { + desc.ViewDimension = Direct3D12::D3D12_SRV_DIMENSION_TEXTURE2DMS; + desc.Anonymous.Texture2DMS = Direct3D12::D3D12_TEX2DMS_SRV { + UnusedField_NothingToDefine: 0, + } + } + wgt::TextureViewDimension::D2 if self.array_layer_base == 0 => { + desc.ViewDimension = Direct3D12::D3D12_SRV_DIMENSION_TEXTURE2D; + desc.Anonymous.Texture2D = Direct3D12::D3D12_TEX2D_SRV { + MostDetailedMip: self.mip_level_base, + MipLevels: self.mip_level_count, + PlaneSlice: aspects_to_plane(self.aspects), + ResourceMinLODClamp: 0.0, + } + } + wgt::TextureViewDimension::D2 | wgt::TextureViewDimension::D2Array + if self.multisampled => + { + desc.ViewDimension = Direct3D12::D3D12_SRV_DIMENSION_TEXTURE2DMSARRAY; + desc.Anonymous.Texture2DMSArray = Direct3D12::D3D12_TEX2DMS_ARRAY_SRV { + FirstArraySlice: self.array_layer_base, + ArraySize: self.array_layer_count, + } + } + wgt::TextureViewDimension::D2 | wgt::TextureViewDimension::D2Array => { + desc.ViewDimension = Direct3D12::D3D12_SRV_DIMENSION_TEXTURE2DARRAY; + desc.Anonymous.Texture2DArray = Direct3D12::D3D12_TEX2D_ARRAY_SRV { + MostDetailedMip: self.mip_level_base, + MipLevels: self.mip_level_count, + FirstArraySlice: self.array_layer_base, + ArraySize: self.array_layer_count, + PlaneSlice: aspects_to_plane(self.aspects), + ResourceMinLODClamp: 0.0, + } + } + wgt::TextureViewDimension::D3 => { + desc.ViewDimension = Direct3D12::D3D12_SRV_DIMENSION_TEXTURE3D; + desc.Anonymous.Texture3D = Direct3D12::D3D12_TEX3D_SRV { + MostDetailedMip: self.mip_level_base, + MipLevels: self.mip_level_count, + ResourceMinLODClamp: 0.0, + } + } + wgt::TextureViewDimension::Cube if self.array_layer_base == 0 => { + desc.ViewDimension = Direct3D12::D3D12_SRV_DIMENSION_TEXTURECUBE; + desc.Anonymous.TextureCube = Direct3D12::D3D12_TEXCUBE_SRV { + MostDetailedMip: self.mip_level_base, + MipLevels: self.mip_level_count, + ResourceMinLODClamp: 0.0, + } + } + wgt::TextureViewDimension::Cube | wgt::TextureViewDimension::CubeArray => { + desc.ViewDimension = Direct3D12::D3D12_SRV_DIMENSION_TEXTURECUBEARRAY; + desc.Anonymous.TextureCubeArray = Direct3D12::D3D12_TEXCUBE_ARRAY_SRV { + MostDetailedMip: self.mip_level_base, + MipLevels: self.mip_level_count, + First2DArrayFace: self.array_layer_base, + NumCubes: if self.array_layer_count == !0 { + !0 + } else { + self.array_layer_count / 6 + }, + ResourceMinLODClamp: 0.0, + } + } + } + + Some(desc) + } + + pub(crate) unsafe fn to_uav(&self) -> Option { + let mut desc = Direct3D12::D3D12_UNORDERED_ACCESS_VIEW_DESC { + Format: self.srv_uav_format?, + ViewDimension: Direct3D12::D3D12_UAV_DIMENSION_UNKNOWN, + Anonymous: Default::default(), + }; + + match self.dimension { + wgt::TextureViewDimension::D1 => { + desc.ViewDimension = Direct3D12::D3D12_UAV_DIMENSION_TEXTURE1D; + desc.Anonymous.Texture1D = Direct3D12::D3D12_TEX1D_UAV { + MipSlice: self.mip_level_base, + } + } + /* + wgt::TextureViewDimension::D1Array => { + desc.ViewDimension = Direct3D12::D3D12_UAV_DIMENSION_TEXTURE1DARRAY; + desc.Anonymous.Texture1DArray = Direct3D12::D3D12_TEX1D_ARRAY_UAV { + MipSlice: self.mip_level_base, + FirstArraySlice: self.array_layer_base, + ArraySize, + } + }*/ + wgt::TextureViewDimension::D2 if self.array_layer_base == 0 => { + desc.ViewDimension = Direct3D12::D3D12_UAV_DIMENSION_TEXTURE2D; + desc.Anonymous.Texture2D = Direct3D12::D3D12_TEX2D_UAV { + MipSlice: self.mip_level_base, + PlaneSlice: aspects_to_plane(self.aspects), + } + } + wgt::TextureViewDimension::D2 | wgt::TextureViewDimension::D2Array => { + desc.ViewDimension = Direct3D12::D3D12_UAV_DIMENSION_TEXTURE2DARRAY; + desc.Anonymous.Texture2DArray = Direct3D12::D3D12_TEX2D_ARRAY_UAV { + MipSlice: self.mip_level_base, + FirstArraySlice: self.array_layer_base, + ArraySize: self.array_layer_count, + PlaneSlice: aspects_to_plane(self.aspects), + } + } + wgt::TextureViewDimension::D3 => { + desc.ViewDimension = Direct3D12::D3D12_UAV_DIMENSION_TEXTURE3D; + desc.Anonymous.Texture3D = Direct3D12::D3D12_TEX3D_UAV { + MipSlice: self.mip_level_base, + FirstWSlice: 0, + WSize: u32::MAX, + } + } + wgt::TextureViewDimension::Cube | wgt::TextureViewDimension::CubeArray => { + panic!("Unable to view texture as cube UAV") + } + } + + Some(desc) + } + + pub(crate) unsafe fn to_rtv(&self) -> Direct3D12::D3D12_RENDER_TARGET_VIEW_DESC { + let mut desc = Direct3D12::D3D12_RENDER_TARGET_VIEW_DESC { + Format: self.rtv_dsv_format, + ViewDimension: Direct3D12::D3D12_RTV_DIMENSION_UNKNOWN, + Anonymous: Default::default(), + }; + + match self.dimension { + wgt::TextureViewDimension::D1 => { + desc.ViewDimension = Direct3D12::D3D12_RTV_DIMENSION_TEXTURE1D; + desc.Anonymous.Texture1D = Direct3D12::D3D12_TEX1D_RTV { + MipSlice: self.mip_level_base, + } + } + /* + wgt::TextureViewDimension::D1Array => { + desc.ViewDimension = Direct3D12::D3D12_RTV_DIMENSION_TEXTURE1DARRAY; + desc.Anonymous.Texture1DArray = Direct3D12::D3D12_TEX1D_ARRAY_RTV { + MipSlice: self.mip_level_base, + FirstArraySlice: self.array_layer_base, + ArraySize, + } + }*/ + wgt::TextureViewDimension::D2 if self.multisampled && self.array_layer_base == 0 => { + desc.ViewDimension = Direct3D12::D3D12_RTV_DIMENSION_TEXTURE2DMS; + desc.Anonymous.Texture2DMS = Direct3D12::D3D12_TEX2DMS_RTV { + UnusedField_NothingToDefine: 0, + } + } + wgt::TextureViewDimension::D2 if self.array_layer_base == 0 => { + desc.ViewDimension = Direct3D12::D3D12_RTV_DIMENSION_TEXTURE2D; + desc.Anonymous.Texture2D = Direct3D12::D3D12_TEX2D_RTV { + MipSlice: self.mip_level_base, + PlaneSlice: aspects_to_plane(self.aspects), + } + } + wgt::TextureViewDimension::D2 | wgt::TextureViewDimension::D2Array + if self.multisampled => + { + desc.ViewDimension = Direct3D12::D3D12_RTV_DIMENSION_TEXTURE2DMSARRAY; + desc.Anonymous.Texture2DMSArray = Direct3D12::D3D12_TEX2DMS_ARRAY_RTV { + FirstArraySlice: self.array_layer_base, + ArraySize: self.array_layer_count, + } + } + wgt::TextureViewDimension::D2 | wgt::TextureViewDimension::D2Array => { + desc.ViewDimension = Direct3D12::D3D12_RTV_DIMENSION_TEXTURE2DARRAY; + desc.Anonymous.Texture2DArray = Direct3D12::D3D12_TEX2D_ARRAY_RTV { + MipSlice: self.mip_level_base, + FirstArraySlice: self.array_layer_base, + ArraySize: self.array_layer_count, + PlaneSlice: aspects_to_plane(self.aspects), + } + } + wgt::TextureViewDimension::D3 + | wgt::TextureViewDimension::Cube + | wgt::TextureViewDimension::CubeArray => { + panic!("Unable to view texture as cube or 3D RTV") + } + } + + desc + } + + pub(crate) unsafe fn to_dsv( + &self, + read_only: bool, + ) -> Direct3D12::D3D12_DEPTH_STENCIL_VIEW_DESC { + let mut desc = Direct3D12::D3D12_DEPTH_STENCIL_VIEW_DESC { + Format: self.rtv_dsv_format, + ViewDimension: Direct3D12::D3D12_DSV_DIMENSION_UNKNOWN, + Flags: { + let mut flags = Direct3D12::D3D12_DSV_FLAG_NONE; + if read_only { + if self.aspects.contains(crate::FormatAspects::DEPTH) { + flags |= Direct3D12::D3D12_DSV_FLAG_READ_ONLY_DEPTH; + } + if self.aspects.contains(crate::FormatAspects::STENCIL) { + flags |= Direct3D12::D3D12_DSV_FLAG_READ_ONLY_STENCIL; + } + } + flags + }, + Anonymous: Default::default(), + }; + + match self.dimension { + wgt::TextureViewDimension::D1 => { + desc.ViewDimension = Direct3D12::D3D12_DSV_DIMENSION_TEXTURE1D; + desc.Anonymous.Texture1D = Direct3D12::D3D12_TEX1D_DSV { + MipSlice: self.mip_level_base, + } + } + /* + wgt::TextureViewDimension::D1Array => { + desc.ViewDimension = Direct3D12::D3D12_DSV_DIMENSION_TEXTURE1DARRAY; + desc.Anonymous.Texture1DArray = Direct3D12::D3D12_TEX1D_ARRAY_DSV { + MipSlice: self.mip_level_base, + FirstArraySlice: self.array_layer_base, + ArraySize, + } + }*/ + wgt::TextureViewDimension::D2 if self.multisampled && self.array_layer_base == 0 => { + desc.ViewDimension = Direct3D12::D3D12_DSV_DIMENSION_TEXTURE2DMS; + desc.Anonymous.Texture2DMS = Direct3D12::D3D12_TEX2DMS_DSV { + UnusedField_NothingToDefine: 0, + } + } + wgt::TextureViewDimension::D2 if self.array_layer_base == 0 => { + desc.ViewDimension = Direct3D12::D3D12_DSV_DIMENSION_TEXTURE2D; + + desc.Anonymous.Texture2D = Direct3D12::D3D12_TEX2D_DSV { + MipSlice: self.mip_level_base, + } + } + wgt::TextureViewDimension::D2 | wgt::TextureViewDimension::D2Array + if self.multisampled => + { + desc.ViewDimension = Direct3D12::D3D12_DSV_DIMENSION_TEXTURE2DMSARRAY; + desc.Anonymous.Texture2DMSArray = Direct3D12::D3D12_TEX2DMS_ARRAY_DSV { + FirstArraySlice: self.array_layer_base, + ArraySize: self.array_layer_count, + } + } + wgt::TextureViewDimension::D2 | wgt::TextureViewDimension::D2Array => { + desc.ViewDimension = Direct3D12::D3D12_DSV_DIMENSION_TEXTURE2DARRAY; + desc.Anonymous.Texture2DArray = Direct3D12::D3D12_TEX2D_ARRAY_DSV { + MipSlice: self.mip_level_base, + FirstArraySlice: self.array_layer_base, + ArraySize: self.array_layer_count, + } + } + wgt::TextureViewDimension::D3 + | wgt::TextureViewDimension::Cube + | wgt::TextureViewDimension::CubeArray => { + panic!("Unable to view texture as cube or 3D DSV") + } + } + + desc + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dynamic/adapter.rs b/third_party/wgpu-hal-29.0.4/src/dynamic/adapter.rs new file mode 100644 index 0000000..6256a33 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dynamic/adapter.rs @@ -0,0 +1,81 @@ +use alloc::boxed::Box; + +use crate::{ + Adapter, Api, DeviceError, OpenDevice, SurfaceCapabilities, TextureFormatCapabilities, +}; + +use super::{DynDevice, DynQueue, DynResource, DynResourceExt, DynSurface}; + +pub struct DynOpenDevice { + pub device: Box, + pub queue: Box, +} + +impl From> for DynOpenDevice { + fn from(open_device: OpenDevice) -> Self { + Self { + device: Box::new(open_device.device), + queue: Box::new(open_device.queue), + } + } +} + +pub trait DynAdapter: DynResource { + unsafe fn open( + &self, + features: wgt::Features, + limits: &wgt::Limits, + memory_hints: &wgt::MemoryHints, + ) -> Result; + + unsafe fn texture_format_capabilities( + &self, + format: wgt::TextureFormat, + ) -> TextureFormatCapabilities; + + unsafe fn surface_capabilities(&self, surface: &dyn DynSurface) -> Option; + + unsafe fn get_presentation_timestamp(&self) -> wgt::PresentationTimestamp; + + fn get_ordered_buffer_usages(&self) -> wgt::BufferUses; + + fn get_ordered_texture_usages(&self) -> wgt::TextureUses; +} + +impl DynAdapter for A { + unsafe fn open( + &self, + features: wgt::Features, + limits: &wgt::Limits, + memory_hints: &wgt::MemoryHints, + ) -> Result { + unsafe { A::open(self, features, limits, memory_hints) }.map(|open_device| DynOpenDevice { + device: Box::new(open_device.device), + queue: Box::new(open_device.queue), + }) + } + + unsafe fn texture_format_capabilities( + &self, + format: wgt::TextureFormat, + ) -> TextureFormatCapabilities { + unsafe { A::texture_format_capabilities(self, format) } + } + + unsafe fn surface_capabilities(&self, surface: &dyn DynSurface) -> Option { + let surface = surface.expect_downcast_ref(); + unsafe { A::surface_capabilities(self, surface) } + } + + unsafe fn get_presentation_timestamp(&self) -> wgt::PresentationTimestamp { + unsafe { A::get_presentation_timestamp(self) } + } + + fn get_ordered_buffer_usages(&self) -> wgt::BufferUses { + A::get_ordered_buffer_usages(self) + } + + fn get_ordered_texture_usages(&self) -> wgt::TextureUses { + A::get_ordered_texture_usages(self) + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dynamic/command.rs b/third_party/wgpu-hal-29.0.4/src/dynamic/command.rs new file mode 100644 index 0000000..6c99970 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dynamic/command.rs @@ -0,0 +1,761 @@ +use alloc::{boxed::Box, vec::Vec}; +use core::ops::Range; + +use crate::{ + AccelerationStructureBarrier, Api, Attachment, BufferBarrier, BufferBinding, BufferCopy, + BufferTextureCopy, BuildAccelerationStructureDescriptor, ColorAttachment, CommandEncoder, + ComputePassDescriptor, DepthStencilAttachment, DeviceError, Label, MemoryRange, + PassTimestampWrites, Rect, RenderPassDescriptor, TextureBarrier, TextureCopy, +}; + +use super::{ + DynAccelerationStructure, DynBindGroup, DynBuffer, DynCommandBuffer, DynComputePipeline, + DynPipelineLayout, DynQuerySet, DynRenderPipeline, DynResource, DynResourceExt as _, + DynTexture, DynTextureView, +}; + +pub trait DynCommandEncoder: DynResource + core::fmt::Debug { + unsafe fn begin_encoding(&mut self, label: Label) -> Result<(), DeviceError>; + + unsafe fn discard_encoding(&mut self); + + unsafe fn end_encoding(&mut self) -> Result, DeviceError>; + + unsafe fn reset_all(&mut self, command_buffers: Vec>); + + unsafe fn transition_buffers(&mut self, barriers: &[BufferBarrier<'_, dyn DynBuffer>]); + unsafe fn transition_textures(&mut self, barriers: &[TextureBarrier<'_, dyn DynTexture>]); + + unsafe fn clear_buffer(&mut self, buffer: &dyn DynBuffer, range: MemoryRange); + + unsafe fn copy_buffer_to_buffer( + &mut self, + src: &dyn DynBuffer, + dst: &dyn DynBuffer, + regions: &[BufferCopy], + ); + + unsafe fn copy_texture_to_texture( + &mut self, + src: &dyn DynTexture, + src_usage: wgt::TextureUses, + dst: &dyn DynTexture, + regions: &[TextureCopy], + ); + + unsafe fn copy_buffer_to_texture( + &mut self, + src: &dyn DynBuffer, + dst: &dyn DynTexture, + regions: &[BufferTextureCopy], + ); + + unsafe fn copy_texture_to_buffer( + &mut self, + src: &dyn DynTexture, + src_usage: wgt::TextureUses, + dst: &dyn DynBuffer, + regions: &[BufferTextureCopy], + ); + + unsafe fn set_bind_group( + &mut self, + layout: &dyn DynPipelineLayout, + index: u32, + group: &dyn DynBindGroup, + dynamic_offsets: &[wgt::DynamicOffset], + ); + + unsafe fn set_immediates( + &mut self, + layout: &dyn DynPipelineLayout, + offset_bytes: u32, + data: &[u32], + ); + + unsafe fn insert_debug_marker(&mut self, label: &str); + unsafe fn begin_debug_marker(&mut self, group_label: &str); + unsafe fn end_debug_marker(&mut self); + + unsafe fn begin_query(&mut self, set: &dyn DynQuerySet, index: u32); + unsafe fn end_query(&mut self, set: &dyn DynQuerySet, index: u32); + unsafe fn write_timestamp(&mut self, set: &dyn DynQuerySet, index: u32); + unsafe fn reset_queries(&mut self, set: &dyn DynQuerySet, range: Range); + unsafe fn copy_query_results( + &mut self, + set: &dyn DynQuerySet, + range: Range, + buffer: &dyn DynBuffer, + offset: wgt::BufferAddress, + stride: wgt::BufferSize, + ); + + unsafe fn begin_render_pass( + &mut self, + desc: &RenderPassDescriptor, + ) -> Result<(), DeviceError>; + unsafe fn end_render_pass(&mut self); + + unsafe fn set_render_pipeline(&mut self, pipeline: &dyn DynRenderPipeline); + + unsafe fn set_index_buffer<'a>( + &mut self, + binding: BufferBinding<'a, dyn DynBuffer>, + format: wgt::IndexFormat, + ); + + unsafe fn set_vertex_buffer<'a>( + &mut self, + index: u32, + binding: BufferBinding<'a, dyn DynBuffer>, + ); + unsafe fn set_viewport(&mut self, rect: &Rect, depth_range: Range); + unsafe fn set_scissor_rect(&mut self, rect: &Rect); + unsafe fn set_stencil_reference(&mut self, value: u32); + unsafe fn set_blend_constants(&mut self, color: &[f32; 4]); + + unsafe fn draw( + &mut self, + first_vertex: u32, + vertex_count: u32, + first_instance: u32, + instance_count: u32, + ); + unsafe fn draw_indexed( + &mut self, + first_index: u32, + index_count: u32, + base_vertex: i32, + first_instance: u32, + instance_count: u32, + ); + unsafe fn draw_mesh_tasks( + &mut self, + group_count_x: u32, + group_count_y: u32, + group_count_z: u32, + ); + unsafe fn draw_indirect( + &mut self, + buffer: &dyn DynBuffer, + offset: wgt::BufferAddress, + draw_count: u32, + ); + unsafe fn draw_indexed_indirect( + &mut self, + buffer: &dyn DynBuffer, + offset: wgt::BufferAddress, + draw_count: u32, + ); + unsafe fn draw_mesh_tasks_indirect( + &mut self, + buffer: &dyn DynBuffer, + offset: wgt::BufferAddress, + draw_count: u32, + ); + unsafe fn draw_indirect_count( + &mut self, + buffer: &dyn DynBuffer, + offset: wgt::BufferAddress, + count_buffer: &dyn DynBuffer, + count_offset: wgt::BufferAddress, + max_count: u32, + ); + unsafe fn draw_indexed_indirect_count( + &mut self, + buffer: &dyn DynBuffer, + offset: wgt::BufferAddress, + count_buffer: &dyn DynBuffer, + count_offset: wgt::BufferAddress, + max_count: u32, + ); + unsafe fn draw_mesh_tasks_indirect_count( + &mut self, + buffer: &dyn DynBuffer, + offset: wgt::BufferAddress, + count_buffer: &dyn DynBuffer, + count_offset: wgt::BufferAddress, + max_count: u32, + ); + + unsafe fn begin_compute_pass(&mut self, desc: &ComputePassDescriptor); + unsafe fn end_compute_pass(&mut self); + + unsafe fn set_compute_pipeline(&mut self, pipeline: &dyn DynComputePipeline); + + unsafe fn dispatch(&mut self, count: [u32; 3]); + unsafe fn dispatch_indirect(&mut self, buffer: &dyn DynBuffer, offset: wgt::BufferAddress); + + unsafe fn build_acceleration_structures<'a>( + &mut self, + descriptors: &'a [BuildAccelerationStructureDescriptor< + 'a, + dyn DynBuffer, + dyn DynAccelerationStructure, + >], + ); + unsafe fn place_acceleration_structure_barrier( + &mut self, + barrier: AccelerationStructureBarrier, + ); + unsafe fn copy_acceleration_structure_to_acceleration_structure( + &mut self, + src: &dyn DynAccelerationStructure, + dst: &dyn DynAccelerationStructure, + copy: wgt::AccelerationStructureCopy, + ); + unsafe fn read_acceleration_structure_compact_size( + &mut self, + acceleration_structure: &dyn DynAccelerationStructure, + buf: &dyn DynBuffer, + ); + unsafe fn set_acceleration_structure_dependencies( + &self, + command_buffers: &[Box], + dependencies: &[&dyn DynAccelerationStructure], + ); +} + +impl DynCommandEncoder for C { + unsafe fn begin_encoding(&mut self, label: Label) -> Result<(), DeviceError> { + unsafe { C::begin_encoding(self, label) } + } + + unsafe fn discard_encoding(&mut self) { + unsafe { C::discard_encoding(self) } + } + + unsafe fn end_encoding(&mut self) -> Result, DeviceError> { + unsafe { C::end_encoding(self) }.map(|cb| { + let boxed_command_buffer: Box<::CommandBuffer> = Box::new(cb); + let boxed_command_buffer: Box = boxed_command_buffer; + boxed_command_buffer + }) + } + + unsafe fn reset_all(&mut self, command_buffers: Vec>) { + unsafe { C::reset_all(self, command_buffers.into_iter().map(|cb| cb.unbox())) } + } + + unsafe fn transition_buffers(&mut self, barriers: &[BufferBarrier<'_, dyn DynBuffer>]) { + let barriers = barriers.iter().map(|barrier| BufferBarrier { + buffer: barrier.buffer.expect_downcast_ref(), + usage: barrier.usage.clone(), + }); + unsafe { self.transition_buffers(barriers) }; + } + + unsafe fn transition_textures(&mut self, barriers: &[TextureBarrier<'_, dyn DynTexture>]) { + let barriers = barriers.iter().map(|barrier| TextureBarrier { + texture: barrier.texture.expect_downcast_ref(), + usage: barrier.usage.clone(), + range: barrier.range, + }); + unsafe { self.transition_textures(barriers) }; + } + + unsafe fn clear_buffer(&mut self, buffer: &dyn DynBuffer, range: MemoryRange) { + let buffer = buffer.expect_downcast_ref(); + unsafe { C::clear_buffer(self, buffer, range) }; + } + + unsafe fn copy_buffer_to_buffer( + &mut self, + src: &dyn DynBuffer, + dst: &dyn DynBuffer, + regions: &[BufferCopy], + ) { + let src = src.expect_downcast_ref(); + let dst = dst.expect_downcast_ref(); + unsafe { + C::copy_buffer_to_buffer(self, src, dst, regions.iter().copied()); + } + } + + unsafe fn copy_texture_to_texture( + &mut self, + src: &dyn DynTexture, + src_usage: wgt::TextureUses, + dst: &dyn DynTexture, + regions: &[TextureCopy], + ) { + let src = src.expect_downcast_ref(); + let dst = dst.expect_downcast_ref(); + unsafe { + C::copy_texture_to_texture(self, src, src_usage, dst, regions.iter().cloned()); + } + } + + unsafe fn copy_buffer_to_texture( + &mut self, + src: &dyn DynBuffer, + dst: &dyn DynTexture, + regions: &[BufferTextureCopy], + ) { + let src = src.expect_downcast_ref(); + let dst = dst.expect_downcast_ref(); + unsafe { + C::copy_buffer_to_texture(self, src, dst, regions.iter().cloned()); + } + } + + unsafe fn copy_texture_to_buffer( + &mut self, + src: &dyn DynTexture, + src_usage: wgt::TextureUses, + dst: &dyn DynBuffer, + regions: &[BufferTextureCopy], + ) { + let src = src.expect_downcast_ref(); + let dst = dst.expect_downcast_ref(); + unsafe { + C::copy_texture_to_buffer(self, src, src_usage, dst, regions.iter().cloned()); + } + } + + unsafe fn set_bind_group( + &mut self, + layout: &dyn DynPipelineLayout, + index: u32, + group: &dyn DynBindGroup, + dynamic_offsets: &[wgt::DynamicOffset], + ) { + let layout = layout.expect_downcast_ref(); + let group = group.expect_downcast_ref(); + unsafe { C::set_bind_group(self, layout, index, group, dynamic_offsets) }; + } + + unsafe fn set_immediates( + &mut self, + layout: &dyn DynPipelineLayout, + offset_bytes: u32, + data: &[u32], + ) { + let layout = layout.expect_downcast_ref(); + unsafe { C::set_immediates(self, layout, offset_bytes, data) }; + } + + unsafe fn insert_debug_marker(&mut self, label: &str) { + unsafe { + C::insert_debug_marker(self, label); + } + } + + unsafe fn begin_debug_marker(&mut self, group_label: &str) { + unsafe { + C::begin_debug_marker(self, group_label); + } + } + + unsafe fn end_debug_marker(&mut self) { + unsafe { + C::end_debug_marker(self); + } + } + + unsafe fn begin_query(&mut self, set: &dyn DynQuerySet, index: u32) { + let set = set.expect_downcast_ref(); + unsafe { C::begin_query(self, set, index) }; + } + + unsafe fn end_query(&mut self, set: &dyn DynQuerySet, index: u32) { + let set = set.expect_downcast_ref(); + unsafe { C::end_query(self, set, index) }; + } + + unsafe fn write_timestamp(&mut self, set: &dyn DynQuerySet, index: u32) { + let set = set.expect_downcast_ref(); + unsafe { C::write_timestamp(self, set, index) }; + } + + unsafe fn reset_queries(&mut self, set: &dyn DynQuerySet, range: Range) { + let set = set.expect_downcast_ref(); + unsafe { C::reset_queries(self, set, range) }; + } + + unsafe fn copy_query_results( + &mut self, + set: &dyn DynQuerySet, + range: Range, + buffer: &dyn DynBuffer, + offset: wgt::BufferAddress, + stride: wgt::BufferSize, + ) { + let set = set.expect_downcast_ref(); + let buffer = buffer.expect_downcast_ref(); + unsafe { C::copy_query_results(self, set, range, buffer, offset, stride) }; + } + + unsafe fn begin_render_pass( + &mut self, + desc: &RenderPassDescriptor, + ) -> Result<(), DeviceError> { + let color_attachments = desc + .color_attachments + .iter() + .map(|attachment| { + attachment + .as_ref() + .map(|attachment| attachment.expect_downcast()) + }) + .collect::>(); + + let desc: RenderPassDescriptor<::QuerySet, ::TextureView> = + RenderPassDescriptor { + label: desc.label, + extent: desc.extent, + sample_count: desc.sample_count, + color_attachments: &color_attachments, + depth_stencil_attachment: desc + .depth_stencil_attachment + .as_ref() + .map(|ds| ds.expect_downcast()), + multiview_mask: desc.multiview_mask, + timestamp_writes: desc + .timestamp_writes + .as_ref() + .map(|writes| writes.expect_downcast()), + occlusion_query_set: desc + .occlusion_query_set + .map(|set| set.expect_downcast_ref()), + }; + unsafe { C::begin_render_pass(self, &desc) } + } + + unsafe fn end_render_pass(&mut self) { + unsafe { + C::end_render_pass(self); + } + } + + unsafe fn set_viewport(&mut self, rect: &Rect, depth_range: Range) { + unsafe { + C::set_viewport(self, rect, depth_range); + } + } + + unsafe fn set_scissor_rect(&mut self, rect: &Rect) { + unsafe { + C::set_scissor_rect(self, rect); + } + } + + unsafe fn set_stencil_reference(&mut self, value: u32) { + unsafe { + C::set_stencil_reference(self, value); + } + } + + unsafe fn set_blend_constants(&mut self, color: &[f32; 4]) { + unsafe { C::set_blend_constants(self, color) }; + } + + unsafe fn draw( + &mut self, + first_vertex: u32, + vertex_count: u32, + first_instance: u32, + instance_count: u32, + ) { + unsafe { + C::draw( + self, + first_vertex, + vertex_count, + first_instance, + instance_count, + ) + }; + } + + unsafe fn draw_indexed( + &mut self, + first_index: u32, + index_count: u32, + base_vertex: i32, + first_instance: u32, + instance_count: u32, + ) { + unsafe { + C::draw_indexed( + self, + first_index, + index_count, + base_vertex, + first_instance, + instance_count, + ) + }; + } + + unsafe fn draw_mesh_tasks( + &mut self, + group_count_x: u32, + group_count_y: u32, + group_count_z: u32, + ) { + unsafe { C::draw_mesh_tasks(self, group_count_x, group_count_y, group_count_z) }; + } + + unsafe fn draw_indirect( + &mut self, + buffer: &dyn DynBuffer, + offset: wgt::BufferAddress, + draw_count: u32, + ) { + let buffer = buffer.expect_downcast_ref(); + unsafe { C::draw_indirect(self, buffer, offset, draw_count) }; + } + + unsafe fn draw_indexed_indirect( + &mut self, + buffer: &dyn DynBuffer, + offset: wgt::BufferAddress, + draw_count: u32, + ) { + let buffer = buffer.expect_downcast_ref(); + unsafe { C::draw_indexed_indirect(self, buffer, offset, draw_count) }; + } + + unsafe fn draw_mesh_tasks_indirect( + &mut self, + buffer: &dyn DynBuffer, + offset: wgt::BufferAddress, + draw_count: u32, + ) { + let buffer = buffer.expect_downcast_ref(); + unsafe { C::draw_mesh_tasks_indirect(self, buffer, offset, draw_count) }; + } + + unsafe fn draw_indirect_count( + &mut self, + buffer: &dyn DynBuffer, + offset: wgt::BufferAddress, + count_buffer: &dyn DynBuffer, + count_offset: wgt::BufferAddress, + max_count: u32, + ) { + let buffer = buffer.expect_downcast_ref(); + let count_buffer = count_buffer.expect_downcast_ref(); + unsafe { + C::draw_indirect_count(self, buffer, offset, count_buffer, count_offset, max_count) + }; + } + + unsafe fn draw_indexed_indirect_count( + &mut self, + buffer: &dyn DynBuffer, + offset: wgt::BufferAddress, + count_buffer: &dyn DynBuffer, + count_offset: wgt::BufferAddress, + max_count: u32, + ) { + let buffer = buffer.expect_downcast_ref(); + let count_buffer = count_buffer.expect_downcast_ref(); + unsafe { + C::draw_indexed_indirect_count( + self, + buffer, + offset, + count_buffer, + count_offset, + max_count, + ) + }; + } + + unsafe fn draw_mesh_tasks_indirect_count( + &mut self, + buffer: &dyn DynBuffer, + offset: wgt::BufferAddress, + count_buffer: &dyn DynBuffer, + count_offset: wgt::BufferAddress, + max_count: u32, + ) { + let buffer = buffer.expect_downcast_ref(); + let count_buffer = count_buffer.expect_downcast_ref(); + unsafe { + C::draw_mesh_tasks_indirect_count( + self, + buffer, + offset, + count_buffer, + count_offset, + max_count, + ) + }; + } + + unsafe fn begin_compute_pass(&mut self, desc: &ComputePassDescriptor) { + let desc = ComputePassDescriptor { + label: desc.label, + timestamp_writes: desc + .timestamp_writes + .as_ref() + .map(|writes| writes.expect_downcast()), + }; + unsafe { C::begin_compute_pass(self, &desc) }; + } + + unsafe fn end_compute_pass(&mut self) { + unsafe { C::end_compute_pass(self) }; + } + + unsafe fn set_compute_pipeline(&mut self, pipeline: &dyn DynComputePipeline) { + let pipeline = pipeline.expect_downcast_ref(); + unsafe { C::set_compute_pipeline(self, pipeline) }; + } + + unsafe fn dispatch(&mut self, count: [u32; 3]) { + unsafe { C::dispatch(self, count) }; + } + + unsafe fn dispatch_indirect(&mut self, buffer: &dyn DynBuffer, offset: wgt::BufferAddress) { + let buffer = buffer.expect_downcast_ref(); + unsafe { C::dispatch_indirect(self, buffer, offset) }; + } + + unsafe fn set_render_pipeline(&mut self, pipeline: &dyn DynRenderPipeline) { + let pipeline = pipeline.expect_downcast_ref(); + unsafe { C::set_render_pipeline(self, pipeline) }; + } + + unsafe fn set_index_buffer<'a>( + &mut self, + binding: BufferBinding<'a, dyn DynBuffer>, + format: wgt::IndexFormat, + ) { + let binding = binding.expect_downcast(); + unsafe { self.set_index_buffer(binding, format) }; + } + + unsafe fn set_vertex_buffer<'a>( + &mut self, + index: u32, + binding: BufferBinding<'a, dyn DynBuffer>, + ) { + let binding = binding.expect_downcast(); + unsafe { self.set_vertex_buffer(index, binding) }; + } + + unsafe fn build_acceleration_structures<'a>( + &mut self, + descriptors: &'a [BuildAccelerationStructureDescriptor< + 'a, + dyn DynBuffer, + dyn DynAccelerationStructure, + >], + ) { + // Need to collect entries here so we can reference them in the descriptor. + // TODO: API should be redesigned to avoid this and other descriptor copies that happen due to the dyn api. + let descriptor_entries = descriptors + .iter() + .map(|d| d.entries.expect_downcast()) + .collect::>(); + let descriptors = descriptors + .iter() + .zip(descriptor_entries.iter()) + .map(|(d, entries)| BuildAccelerationStructureDescriptor::< + ::Buffer, + ::AccelerationStructure, + > { + entries, + mode: d.mode, + flags: d.flags, + source_acceleration_structure: d + .source_acceleration_structure + .map(|a| a.expect_downcast_ref()), + destination_acceleration_structure: d + .destination_acceleration_structure + .expect_downcast_ref(), + scratch_buffer: d.scratch_buffer.expect_downcast_ref(), + scratch_buffer_offset: d.scratch_buffer_offset, + }); + unsafe { C::build_acceleration_structures(self, descriptors.len() as _, descriptors) }; + } + + unsafe fn place_acceleration_structure_barrier( + &mut self, + barrier: AccelerationStructureBarrier, + ) { + unsafe { C::place_acceleration_structure_barrier(self, barrier) }; + } + + unsafe fn copy_acceleration_structure_to_acceleration_structure( + &mut self, + src: &dyn DynAccelerationStructure, + dst: &dyn DynAccelerationStructure, + copy: wgt::AccelerationStructureCopy, + ) { + let src = src.expect_downcast_ref(); + let dst = dst.expect_downcast_ref(); + unsafe { C::copy_acceleration_structure_to_acceleration_structure(self, src, dst, copy) }; + } + unsafe fn read_acceleration_structure_compact_size( + &mut self, + acceleration_structure: &dyn DynAccelerationStructure, + buf: &dyn DynBuffer, + ) { + let acceleration_structure = acceleration_structure.expect_downcast_ref(); + let buf = buf.expect_downcast_ref(); + unsafe { C::read_acceleration_structure_compact_size(self, acceleration_structure, buf) } + } + + unsafe fn set_acceleration_structure_dependencies( + &self, + command_buffers: &[Box], + dependencies: &[&dyn DynAccelerationStructure], + ) { + let command_buffers: Vec<&::CommandBuffer> = command_buffers + .iter() + .map(|command_buffer| command_buffer.expect_downcast_ref()) + .collect(); + let dependencies: Vec<&::AccelerationStructure> = dependencies + .iter() + .map(|dependency| dependency.expect_downcast_ref()) + .collect(); + unsafe { C::set_acceleration_structure_dependencies(&command_buffers, &dependencies) } + } +} + +impl<'a> PassTimestampWrites<'a, dyn DynQuerySet> { + pub fn expect_downcast(&self) -> PassTimestampWrites<'a, B> { + PassTimestampWrites { + query_set: self.query_set.expect_downcast_ref(), + beginning_of_pass_write_index: self.beginning_of_pass_write_index, + end_of_pass_write_index: self.end_of_pass_write_index, + } + } +} + +impl<'a> Attachment<'a, dyn DynTextureView> { + pub fn expect_downcast(&self) -> Attachment<'a, B> { + Attachment { + view: self.view.expect_downcast_ref(), + usage: self.usage, + } + } +} + +impl<'a> ColorAttachment<'a, dyn DynTextureView> { + pub fn expect_downcast(&self) -> ColorAttachment<'a, B> { + ColorAttachment { + target: self.target.expect_downcast(), + depth_slice: self.depth_slice, + resolve_target: self.resolve_target.as_ref().map(|rt| rt.expect_downcast()), + ops: self.ops, + clear_value: self.clear_value, + } + } +} + +impl<'a> DepthStencilAttachment<'a, dyn DynTextureView> { + pub fn expect_downcast(&self) -> DepthStencilAttachment<'a, B> { + DepthStencilAttachment { + target: self.target.expect_downcast(), + depth_ops: self.depth_ops, + stencil_ops: self.stencil_ops, + clear_value: self.clear_value, + } + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dynamic/device.rs b/third_party/wgpu-hal-29.0.4/src/dynamic/device.rs new file mode 100644 index 0000000..b5ff590 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dynamic/device.rs @@ -0,0 +1,558 @@ +use alloc::{borrow::ToOwned as _, boxed::Box, vec::Vec}; + +use crate::{ + AccelerationStructureBuildSizes, AccelerationStructureDescriptor, Api, BindGroupDescriptor, + BindGroupLayoutDescriptor, BufferDescriptor, BufferMapping, CommandEncoderDescriptor, + ComputePipelineDescriptor, Device, DeviceError, FenceValue, + GetAccelerationStructureBuildSizesDescriptor, Label, MemoryRange, PipelineCacheDescriptor, + PipelineCacheError, PipelineError, PipelineLayoutDescriptor, RenderPipelineDescriptor, + SamplerDescriptor, ShaderError, ShaderInput, ShaderModuleDescriptor, TextureDescriptor, + TextureViewDescriptor, TlasInstance, +}; + +use super::{ + DynAccelerationStructure, DynBindGroup, DynBindGroupLayout, DynBuffer, DynCommandEncoder, + DynComputePipeline, DynFence, DynPipelineCache, DynPipelineLayout, DynQuerySet, DynQueue, + DynRenderPipeline, DynResource, DynResourceExt as _, DynSampler, DynShaderModule, DynTexture, + DynTextureView, +}; + +pub trait DynDevice: DynResource { + unsafe fn create_buffer( + &self, + desc: &BufferDescriptor, + ) -> Result, DeviceError>; + + unsafe fn destroy_buffer(&self, buffer: Box); + unsafe fn add_raw_buffer(&self, buffer: &dyn DynBuffer); + + unsafe fn map_buffer( + &self, + buffer: &dyn DynBuffer, + range: MemoryRange, + ) -> Result; + + unsafe fn unmap_buffer(&self, buffer: &dyn DynBuffer); + + unsafe fn flush_mapped_ranges(&self, buffer: &dyn DynBuffer, ranges: &[MemoryRange]); + unsafe fn invalidate_mapped_ranges(&self, buffer: &dyn DynBuffer, ranges: &[MemoryRange]); + + unsafe fn create_texture( + &self, + desc: &TextureDescriptor, + ) -> Result, DeviceError>; + unsafe fn destroy_texture(&self, texture: Box); + unsafe fn add_raw_texture(&self, texture: &dyn DynTexture); + + unsafe fn create_texture_view( + &self, + texture: &dyn DynTexture, + desc: &TextureViewDescriptor, + ) -> Result, DeviceError>; + unsafe fn destroy_texture_view(&self, view: Box); + unsafe fn create_sampler( + &self, + desc: &SamplerDescriptor, + ) -> Result, DeviceError>; + unsafe fn destroy_sampler(&self, sampler: Box); + + unsafe fn create_command_encoder( + &self, + desc: &CommandEncoderDescriptor, + ) -> Result, DeviceError>; + + unsafe fn create_bind_group_layout( + &self, + desc: &BindGroupLayoutDescriptor, + ) -> Result, DeviceError>; + unsafe fn destroy_bind_group_layout(&self, bg_layout: Box); + + unsafe fn create_pipeline_layout( + &self, + desc: &PipelineLayoutDescriptor, + ) -> Result, DeviceError>; + unsafe fn destroy_pipeline_layout(&self, pipeline_layout: Box); + + unsafe fn create_bind_group( + &self, + desc: &BindGroupDescriptor< + dyn DynBindGroupLayout, + dyn DynBuffer, + dyn DynSampler, + dyn DynTextureView, + dyn DynAccelerationStructure, + >, + ) -> Result, DeviceError>; + unsafe fn destroy_bind_group(&self, group: Box); + + unsafe fn create_shader_module( + &self, + desc: &ShaderModuleDescriptor, + shader: ShaderInput, + ) -> Result, ShaderError>; + unsafe fn destroy_shader_module(&self, module: Box); + + unsafe fn create_render_pipeline( + &self, + desc: &RenderPipelineDescriptor< + dyn DynPipelineLayout, + dyn DynShaderModule, + dyn DynPipelineCache, + >, + ) -> Result, PipelineError>; + unsafe fn destroy_render_pipeline(&self, pipeline: Box); + + unsafe fn create_compute_pipeline( + &self, + desc: &ComputePipelineDescriptor< + dyn DynPipelineLayout, + dyn DynShaderModule, + dyn DynPipelineCache, + >, + ) -> Result, PipelineError>; + unsafe fn destroy_compute_pipeline(&self, pipeline: Box); + + unsafe fn create_pipeline_cache( + &self, + desc: &PipelineCacheDescriptor<'_>, + ) -> Result, PipelineCacheError>; + fn pipeline_cache_validation_key(&self) -> Option<[u8; 16]> { + None + } + unsafe fn destroy_pipeline_cache(&self, cache: Box); + + unsafe fn create_query_set( + &self, + desc: &wgt::QuerySetDescriptor) -> Self { + Self { + adapter: Box::new(exposed_adapter.adapter), + info: exposed_adapter.info, + features: exposed_adapter.features, + capabilities: exposed_adapter.capabilities, + } + } +} + +pub trait DynInstance: DynResource { + unsafe fn create_surface( + &self, + display_handle: raw_window_handle::RawDisplayHandle, + window_handle: raw_window_handle::RawWindowHandle, + ) -> Result, InstanceError>; + + unsafe fn enumerate_adapters( + &self, + surface_hint: Option<&dyn DynSurface>, + ) -> Vec; +} + +impl DynInstance for I { + unsafe fn create_surface( + &self, + display_handle: raw_window_handle::RawDisplayHandle, + window_handle: raw_window_handle::RawWindowHandle, + ) -> Result, InstanceError> { + unsafe { I::create_surface(self, display_handle, window_handle) } + .map(|surface| -> Box { Box::new(surface) }) + } + + unsafe fn enumerate_adapters( + &self, + surface_hint: Option<&dyn DynSurface>, + ) -> Vec { + let surface_hint = surface_hint.map(|s| s.expect_downcast_ref()); + unsafe { I::enumerate_adapters(self, surface_hint) } + .into_iter() + .map(|exposed| DynExposedAdapter { + adapter: Box::new(exposed.adapter), + info: exposed.info, + features: exposed.features, + capabilities: exposed.capabilities, + }) + .collect() + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dynamic/mod.rs b/third_party/wgpu-hal-29.0.4/src/dynamic/mod.rs new file mode 100644 index 0000000..85d8ca0 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dynamic/mod.rs @@ -0,0 +1,221 @@ +mod adapter; +mod command; +mod device; +mod instance; +mod queue; +mod surface; + +pub use adapter::{DynAdapter, DynOpenDevice}; +pub use command::DynCommandEncoder; +pub use device::DynDevice; +pub use instance::{DynExposedAdapter, DynInstance}; +pub use queue::DynQueue; +pub use surface::{DynAcquiredSurfaceTexture, DynSurface}; + +use alloc::boxed::Box; +use core::{ + any::{Any, TypeId}, + fmt, +}; + +use wgt::WasmNotSendSync; + +use crate::{ + AccelerationStructureAABBs, AccelerationStructureEntries, AccelerationStructureInstances, + AccelerationStructureTriangleIndices, AccelerationStructureTriangleTransform, + AccelerationStructureTriangles, BufferBinding, ExternalTextureBinding, ProgrammableStage, + TextureBinding, +}; + +/// Base trait for all resources, allows downcasting via [`Any`]. +pub trait DynResource: Any + WasmNotSendSync + 'static { + fn as_any(&self) -> &dyn Any; + fn as_any_mut(&mut self) -> &mut dyn Any; +} + +/// Utility macro for implementing `DynResource` for a list of types. +macro_rules! impl_dyn_resource { + ($($type:ty),*) => { + $( + impl crate::DynResource for $type { + fn as_any(&self) -> &dyn ::core::any::Any { + self + } + + fn as_any_mut(&mut self) -> &mut dyn ::core::any::Any { + self + } + } + )* + }; +} +pub(crate) use impl_dyn_resource; + +/// Extension trait for `DynResource` used by implementations of various dynamic resource traits. +trait DynResourceExt { + /// # Panics + /// + /// - Panics if `self` is not downcastable to `T`. + fn expect_downcast_ref(&self) -> &T; + /// # Panics + /// + /// - Panics if `self` is not downcastable to `T`. + fn expect_downcast_mut(&mut self) -> &mut T; + + /// Unboxes a `Box` to a concrete type. + /// + /// # Safety + /// + /// - `self` must be the correct concrete type. + unsafe fn unbox(self: Box) -> T; +} + +impl DynResourceExt for R { + fn expect_downcast_ref<'a, T: DynResource>(&'a self) -> &'a T { + self.as_any() + .downcast_ref() + .expect("Resource doesn't have the expected backend type.") + } + + fn expect_downcast_mut<'a, T: DynResource>(&'a mut self) -> &'a mut T { + self.as_any_mut() + .downcast_mut() + .expect("Resource doesn't have the expected backend type.") + } + + unsafe fn unbox(self: Box) -> T { + debug_assert!( + ::type_id(self.as_ref()) == TypeId::of::(), + "Resource doesn't have the expected type, expected {:?}, got {:?}", + TypeId::of::(), + ::type_id(self.as_ref()) + ); + + let casted_ptr = Box::into_raw(self).cast::(); + // SAFETY: This is adheres to the safety contract of `Box::from_raw` because: + // + // - We are casting the value of a previously `Box`ed value, which guarantees: + // - `casted_ptr` is not null. + // - `casted_ptr` is valid for reads and writes, though by itself this does not mean + // valid reads and writes for `T` (read on for that). + // - We don't change the allocator. + // - The contract of `Box::from_raw` requires that an initialized and aligned `T` is stored + // within `casted_ptr`. + *unsafe { Box::from_raw(casted_ptr) } + } +} + +pub trait DynAccelerationStructure: DynResource + fmt::Debug {} +pub trait DynBindGroup: DynResource + fmt::Debug {} +pub trait DynBindGroupLayout: DynResource + fmt::Debug {} +pub trait DynBuffer: DynResource + fmt::Debug {} +pub trait DynCommandBuffer: DynResource + fmt::Debug {} +pub trait DynComputePipeline: DynResource + fmt::Debug {} +pub trait DynFence: DynResource + fmt::Debug {} +pub trait DynPipelineCache: DynResource + fmt::Debug {} +pub trait DynPipelineLayout: DynResource + fmt::Debug {} +pub trait DynQuerySet: DynResource + fmt::Debug {} +pub trait DynRenderPipeline: DynResource + fmt::Debug {} +pub trait DynSampler: DynResource + fmt::Debug {} +pub trait DynShaderModule: DynResource + fmt::Debug {} +pub trait DynSurfaceTexture: + DynResource + core::borrow::Borrow + fmt::Debug +{ +} +pub trait DynTexture: DynResource + fmt::Debug {} +pub trait DynTextureView: DynResource + fmt::Debug {} + +impl<'a> BufferBinding<'a, dyn DynBuffer> { + pub fn expect_downcast(self) -> BufferBinding<'a, B> { + BufferBinding { + buffer: self.buffer.expect_downcast_ref(), + offset: self.offset, + size: self.size, + } + } +} + +impl<'a> TextureBinding<'a, dyn DynTextureView> { + pub fn expect_downcast(self) -> TextureBinding<'a, T> { + TextureBinding { + view: self.view.expect_downcast_ref(), + usage: self.usage, + } + } +} + +impl<'a> ExternalTextureBinding<'a, dyn DynBuffer, dyn DynTextureView> { + pub fn expect_downcast( + self, + ) -> ExternalTextureBinding<'a, B, T> { + let planes = self.planes.map(|plane| plane.expect_downcast()); + let params = self.params.expect_downcast(); + ExternalTextureBinding { planes, params } + } +} + +impl<'a> ProgrammableStage<'a, dyn DynShaderModule> { + fn expect_downcast(self) -> ProgrammableStage<'a, T> { + ProgrammableStage { + module: self.module.expect_downcast_ref(), + entry_point: self.entry_point, + constants: self.constants, + zero_initialize_workgroup_memory: self.zero_initialize_workgroup_memory, + } + } +} + +impl<'a> AccelerationStructureEntries<'a, dyn DynBuffer> { + fn expect_downcast(&self) -> AccelerationStructureEntries<'a, B> { + match self { + AccelerationStructureEntries::Instances(instances) => { + AccelerationStructureEntries::Instances(AccelerationStructureInstances { + buffer: instances.buffer.map(|b| b.expect_downcast_ref()), + offset: instances.offset, + count: instances.count, + }) + } + AccelerationStructureEntries::Triangles(triangles) => { + AccelerationStructureEntries::Triangles( + triangles + .iter() + .map(|t| AccelerationStructureTriangles { + vertex_buffer: t.vertex_buffer.map(|b| b.expect_downcast_ref()), + vertex_format: t.vertex_format, + first_vertex: t.first_vertex, + vertex_count: t.vertex_count, + vertex_stride: t.vertex_stride, + indices: t.indices.as_ref().map(|i| { + AccelerationStructureTriangleIndices { + buffer: i.buffer.map(|b| b.expect_downcast_ref()), + format: i.format, + offset: i.offset, + count: i.count, + } + }), + transform: t.transform.as_ref().map(|t| { + AccelerationStructureTriangleTransform { + buffer: t.buffer.expect_downcast_ref(), + offset: t.offset, + } + }), + flags: t.flags, + }) + .collect(), + ) + } + AccelerationStructureEntries::AABBs(entries) => AccelerationStructureEntries::AABBs( + entries + .iter() + .map(|e| AccelerationStructureAABBs { + buffer: e.buffer.map(|b| b.expect_downcast_ref()), + offset: e.offset, + count: e.count, + stride: e.stride, + flags: e.flags, + }) + .collect(), + ), + } + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dynamic/queue.rs b/third_party/wgpu-hal-29.0.4/src/dynamic/queue.rs new file mode 100644 index 0000000..2af6d1f --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dynamic/queue.rs @@ -0,0 +1,56 @@ +use alloc::{boxed::Box, vec::Vec}; + +use crate::{ + DeviceError, DynCommandBuffer, DynFence, DynResource, DynSurface, DynSurfaceTexture, + FenceValue, Queue, SurfaceError, +}; + +use super::DynResourceExt as _; + +pub trait DynQueue: DynResource { + unsafe fn submit( + &self, + command_buffers: &[&dyn DynCommandBuffer], + surface_textures: &[&dyn DynSurfaceTexture], + signal_fence: (&mut dyn DynFence, FenceValue), + ) -> Result<(), DeviceError>; + unsafe fn present( + &self, + surface: &dyn DynSurface, + texture: Box, + ) -> Result<(), SurfaceError>; + unsafe fn get_timestamp_period(&self) -> f32; +} + +impl DynQueue for Q { + unsafe fn submit( + &self, + command_buffers: &[&dyn DynCommandBuffer], + surface_textures: &[&dyn DynSurfaceTexture], + signal_fence: (&mut dyn DynFence, FenceValue), + ) -> Result<(), DeviceError> { + let command_buffers = command_buffers + .iter() + .map(|cb| (*cb).expect_downcast_ref()) + .collect::>(); + let surface_textures = surface_textures + .iter() + .map(|surface| (*surface).expect_downcast_ref()) + .collect::>(); + let signal_fence = (signal_fence.0.expect_downcast_mut(), signal_fence.1); + unsafe { Q::submit(self, &command_buffers, &surface_textures, signal_fence) } + } + + unsafe fn present( + &self, + surface: &dyn DynSurface, + texture: Box, + ) -> Result<(), SurfaceError> { + let surface = surface.expect_downcast_ref(); + unsafe { Q::present(self, surface, texture.unbox()) } + } + + unsafe fn get_timestamp_period(&self) -> f32 { + unsafe { Q::get_timestamp_period(self) } + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/dynamic/surface.rs b/third_party/wgpu-hal-29.0.4/src/dynamic/surface.rs new file mode 100644 index 0000000..17fa8ee --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/dynamic/surface.rs @@ -0,0 +1,72 @@ +use alloc::boxed::Box; +use core::time::Duration; + +use crate::{ + DynDevice, DynFence, DynResource, DynSurfaceTexture, Surface, SurfaceConfiguration, + SurfaceError, +}; + +use super::DynResourceExt as _; + +#[derive(Debug)] +pub struct DynAcquiredSurfaceTexture { + pub texture: Box, + /// The presentation configuration no longer matches + /// the surface properties exactly, but can still be used to present + /// to the surface successfully. + pub suboptimal: bool, +} + +pub trait DynSurface: DynResource { + unsafe fn configure( + &self, + device: &dyn DynDevice, + config: &SurfaceConfiguration, + ) -> Result<(), SurfaceError>; + + unsafe fn unconfigure(&self, device: &dyn DynDevice); + + unsafe fn acquire_texture( + &self, + timeout: Option, + fence: &dyn DynFence, + ) -> Result; + + unsafe fn discard_texture(&self, texture: Box); +} + +impl DynSurface for S { + unsafe fn configure( + &self, + device: &dyn DynDevice, + config: &SurfaceConfiguration, + ) -> Result<(), SurfaceError> { + let device = device.expect_downcast_ref(); + unsafe { S::configure(self, device, config) } + } + + unsafe fn unconfigure(&self, device: &dyn DynDevice) { + let device = device.expect_downcast_ref(); + unsafe { S::unconfigure(self, device) } + } + + unsafe fn acquire_texture( + &self, + timeout: Option, + fence: &dyn DynFence, + ) -> Result { + let fence = fence.expect_downcast_ref(); + unsafe { S::acquire_texture(self, timeout, fence) }.map(|ast| { + let texture = Box::new(ast.texture); + let suboptimal = ast.suboptimal; + DynAcquiredSurfaceTexture { + texture, + suboptimal, + } + }) + } + + unsafe fn discard_texture(&self, texture: Box) { + unsafe { S::discard_texture(self, texture.unbox()) } + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/gles/adapter.rs b/third_party/wgpu-hal-29.0.4/src/gles/adapter.rs new file mode 100644 index 0000000..c682145 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/adapter.rs @@ -0,0 +1,1366 @@ +use alloc::{borrow::ToOwned as _, format, string::String, sync::Arc, vec, vec::Vec}; +use core::sync::atomic::AtomicU8; + +use glow::HasContext; +use parking_lot::Mutex; +use wgt::AstcChannel; + +use crate::auxil::db; +use crate::gles::ShaderClearProgram; + +// https://webgl2fundamentals.org/webgl/lessons/webgl-data-textures.html + +const GL_UNMASKED_VENDOR_WEBGL: u32 = 0x9245; +const GL_UNMASKED_RENDERER_WEBGL: u32 = 0x9246; + +impl super::Adapter { + /// Note that this function is intentionally lenient in regards to parsing, + /// and will try to recover at least the first two version numbers without + /// resulting in an `Err`. + /// # Notes + /// `WebGL 2` version returned as `OpenGL ES 3.0` + fn parse_version(mut src: &str) -> Result<(u8, u8), crate::InstanceError> { + let webgl_sig = "WebGL "; + // According to the WebGL specification + // VERSION WebGL1.0 + // SHADING_LANGUAGE_VERSION WebGLGLSLES1.0 + let is_webgl = src.starts_with(webgl_sig); + if is_webgl { + let pos = src.rfind(webgl_sig).unwrap_or(0); + src = &src[pos + webgl_sig.len()..]; + } else { + let es_sig = " ES "; + match src.rfind(es_sig) { + Some(pos) => { + src = &src[pos + es_sig.len()..]; + } + None => { + return Err(crate::InstanceError::new(format!( + "OpenGL version {src:?} does not contain 'ES'" + ))); + } + } + }; + + let glsl_es_sig = "GLSL ES "; + let is_glsl = match src.find(glsl_es_sig) { + Some(pos) => { + src = &src[pos + glsl_es_sig.len()..]; + true + } + None => false, + }; + + Self::parse_full_version(src).map(|(major, minor)| { + ( + // Return WebGL 2.0 version as OpenGL ES 3.0 + if is_webgl && !is_glsl { + major + 1 + } else { + major + }, + minor, + ) + }) + } + + /// According to the OpenGL specification, the version information is + /// expected to follow the following syntax: + /// + /// ~~~bnf + /// ::= + /// ::= + /// ::= + /// ::= + /// ::= "." ["." ] + /// ::= [" " ] + /// ~~~ + /// + /// Note that this function is intentionally lenient in regards to parsing, + /// and will try to recover at least the first two version numbers without + /// resulting in an `Err`. + pub(super) fn parse_full_version(src: &str) -> Result<(u8, u8), crate::InstanceError> { + let (version, _vendor_info) = match src.find(' ') { + Some(i) => (&src[..i], src[i + 1..].to_owned()), + None => (src, String::new()), + }; + + // TODO: make this even more lenient so that we can also accept + // ` "." []` + let mut it = version.split('.'); + let major = it.next().and_then(|s| s.parse().ok()); + let minor = it.next().and_then(|s| { + let trimmed = if s.starts_with('0') { + "0" + } else { + s.trim_end_matches('0') + }; + trimmed.parse().ok() + }); + + match (major, minor) { + (Some(major), Some(minor)) => Ok((major, minor)), + _ => Err(crate::InstanceError::new(format!( + "unable to extract OpenGL version from {version:?}" + ))), + } + } + + fn make_info(vendor_orig: String, renderer_orig: String, version: String) -> wgt::AdapterInfo { + let vendor = vendor_orig.to_lowercase(); + let renderer = renderer_orig.to_lowercase(); + + // opengl has no way to discern device_type, so we can try to infer it from the renderer string + let strings_that_imply_integrated = [ + " xpress", // space here is on purpose so we don't match express + "amd renoir", + "radeon hd 4200", + "radeon hd 4250", + "radeon hd 4290", + "radeon hd 4270", + "radeon hd 4225", + "radeon hd 3100", + "radeon hd 3200", + "radeon hd 3000", + "radeon hd 3300", + "radeon(tm) r4 graphics", + "radeon(tm) r5 graphics", + "radeon(tm) r6 graphics", + "radeon(tm) r7 graphics", + "radeon r7 graphics", + "nforce", // all nvidia nforce are integrated + "tegra", // all nvidia tegra are integrated + "shield", // all nvidia shield are integrated + "igp", + "mali", + "intel", + "v3d", + "apple m", // all apple m are integrated + ]; + let strings_that_imply_cpu = ["mesa offscreen", "swiftshader", "llvmpipe"]; + + //TODO: handle Intel Iris XE as discreet + let inferred_device_type = if vendor.contains("qualcomm") + || vendor.contains("intel") + || strings_that_imply_integrated + .iter() + .any(|&s| renderer.contains(s)) + { + wgt::DeviceType::IntegratedGpu + } else if strings_that_imply_cpu.iter().any(|&s| renderer.contains(s)) { + wgt::DeviceType::Cpu + } else { + // At this point the Device type is Unknown. + // It's most likely DiscreteGpu, but we do not know for sure. + // Use "Other" to avoid possibly making incorrect assumptions. + // Note that if this same device is available under some other API (ex: Vulkan), + // It will mostly likely get a different device type (probably DiscreteGpu). + wgt::DeviceType::Other + }; + + // source: Sascha Willems at Vulkan + let vendor_id = if vendor.contains("amd") { + db::amd::VENDOR + } else if vendor.contains("imgtec") { + db::imgtec::VENDOR + } else if vendor.contains("nvidia") { + db::nvidia::VENDOR + } else if vendor.contains("arm") { + db::arm::VENDOR + } else if vendor.contains("qualcomm") { + db::qualcomm::VENDOR + } else if vendor.contains("intel") { + db::intel::VENDOR + } else if vendor.contains("broadcom") { + db::broadcom::VENDOR + } else if vendor.contains("mesa") { + db::mesa::VENDOR + } else if vendor.contains("apple") { + db::apple::VENDOR + } else { + 0 + }; + + wgt::AdapterInfo { + name: renderer_orig, + vendor: vendor_id, + device: 0, + device_type: inferred_device_type, + driver: "".to_owned(), + device_pci_bus_id: String::new(), + driver_info: version, + backend: wgt::Backend::Gl, + subgroup_min_size: wgt::MINIMUM_SUBGROUP_MIN_SIZE, + subgroup_max_size: wgt::MAXIMUM_SUBGROUP_MAX_SIZE, + transient_saves_memory: false, + } + } + + pub(super) unsafe fn expose( + context: super::AdapterContext, + backend_options: wgt::GlBackendOptions, + ) -> Option> { + let gl = context.lock(); + let extensions = gl.supported_extensions(); + + let (vendor_const, renderer_const) = if extensions.contains("WEBGL_debug_renderer_info") { + // emscripten doesn't enable "WEBGL_debug_renderer_info" extension by default. so, we do it manually. + // See https://github.com/gfx-rs/wgpu/issues/3245 for context + #[cfg(Emscripten)] + if unsafe { + super::emscripten::enable_extension(c"WEBGL_debug_renderer_info".to_str().unwrap()) + } { + (GL_UNMASKED_VENDOR_WEBGL, GL_UNMASKED_RENDERER_WEBGL) + } else { + (glow::VENDOR, glow::RENDERER) + } + // glow already enables WEBGL_debug_renderer_info on wasm32-unknown-unknown target by default. + #[cfg(not(Emscripten))] + (GL_UNMASKED_VENDOR_WEBGL, GL_UNMASKED_RENDERER_WEBGL) + } else { + (glow::VENDOR, glow::RENDERER) + }; + + let vendor = unsafe { gl.get_parameter_string(vendor_const) }; + let renderer = unsafe { gl.get_parameter_string(renderer_const) }; + let version = unsafe { gl.get_parameter_string(glow::VERSION) }; + log::debug!("Vendor: {vendor}"); + log::debug!("Renderer: {renderer}"); + log::debug!("Version: {version}"); + + let full_ver = Self::parse_full_version(&version).ok(); + let es_ver = full_ver.map_or_else(|| Self::parse_version(&version).ok(), |_| None); + + if let Some(full_ver) = full_ver { + let core_profile = (full_ver >= (3, 2)).then(|| unsafe { + gl.get_parameter_i32(glow::CONTEXT_PROFILE_MASK) + & glow::CONTEXT_CORE_PROFILE_BIT as i32 + != 0 + }); + log::trace!( + "Profile: {}", + core_profile + .map(|core_profile| if core_profile { + "Core" + } else { + "Compatibility" + }) + .unwrap_or("Legacy") + ); + } + + if es_ver.is_none() && full_ver.is_none() { + log::warn!("Unable to parse OpenGL version"); + return None; + } + + if let Some(es_ver) = es_ver { + if es_ver < (3, 0) { + log::warn!( + "Returned GLES context is {}.{}, when 3.0+ was requested", + es_ver.0, + es_ver.1 + ); + return None; + } + } + + if let Some(full_ver) = full_ver { + if full_ver < (3, 3) { + log::warn!( + "Returned GL context is {}.{}, when 3.3+ is needed", + full_ver.0, + full_ver.1 + ); + return None; + } + } + + let shading_language_version = { + let sl_version = unsafe { gl.get_parameter_string(glow::SHADING_LANGUAGE_VERSION) }; + log::debug!("SL version: {}", &sl_version); + if full_ver.is_some() { + let (sl_major, sl_minor) = Self::parse_full_version(&sl_version).ok()?; + let mut value = sl_major as u16 * 100 + sl_minor as u16 * 10; + // Naga doesn't think it supports GL 460+, so we cap it at 450 + if value > 450 { + value = 450; + } + naga::back::glsl::Version::Desktop(value) + } else { + let (sl_major, sl_minor) = Self::parse_version(&sl_version).ok()?; + let value = sl_major as u16 * 100 + sl_minor as u16 * 10; + naga::back::glsl::Version::Embedded { + version: value, + is_webgl: cfg!(any(webgl, Emscripten)), + } + } + }; + + log::debug!("Supported GL Extensions: {extensions:#?}"); + + let supported = |(req_es_major, req_es_minor), (req_full_major, req_full_minor)| { + let es_supported = es_ver + .map(|es_ver| es_ver >= (req_es_major, req_es_minor)) + .unwrap_or_default(); + + let full_supported = full_ver + .map(|full_ver| full_ver >= (req_full_major, req_full_minor)) + .unwrap_or_default(); + + es_supported || full_supported + }; + + let supports_storage = + supported((3, 1), (4, 3)) || extensions.contains("GL_ARB_shader_storage_buffer_object"); + let supports_compute = + supported((3, 1), (4, 3)) || extensions.contains("GL_ARB_compute_shader"); + let supports_work_group_params = supports_compute; + + // ANGLE provides renderer strings like: "ANGLE (Apple, Apple M1 Pro, OpenGL 4.1)" + let is_angle = renderer.contains("ANGLE"); + + let vertex_shader_storage_blocks = if supports_storage { + let value = + (unsafe { gl.get_parameter_i32(glow::MAX_VERTEX_SHADER_STORAGE_BLOCKS) } as u32); + + if value == 0 && extensions.contains("GL_ARB_shader_storage_buffer_object") { + // The driver for AMD Radeon HD 5870 returns zero here, so assume the value matches the compute shader storage block count. + // Windows doesn't recognize `GL_MAX_VERTEX_ATTRIB_STRIDE`. + let new = (unsafe { gl.get_parameter_i32(glow::MAX_COMPUTE_SHADER_STORAGE_BLOCKS) } + as u32); + log::debug!("Max vertex shader storage blocks is zero, but GL_ARB_shader_storage_buffer_object is specified. Assuming the compute value {new}"); + new + } else { + value + } + } else { + 0 + }; + let fragment_shader_storage_blocks = if supports_storage { + (unsafe { gl.get_parameter_i32(glow::MAX_FRAGMENT_SHADER_STORAGE_BLOCKS) } as u32) + } else { + 0 + }; + let vertex_shader_storage_textures = if supports_storage { + (unsafe { gl.get_parameter_i32(glow::MAX_VERTEX_IMAGE_UNIFORMS) } as u32) + } else { + 0 + }; + let fragment_shader_storage_textures = if supports_storage { + (unsafe { gl.get_parameter_i32(glow::MAX_FRAGMENT_IMAGE_UNIFORMS) } as u32) + } else { + 0 + }; + let max_storage_block_size = if supports_storage { + (unsafe { gl.get_parameter_i32(glow::MAX_SHADER_STORAGE_BLOCK_SIZE) } as u32) + } else { + 0 + }; + let max_element_index = unsafe { gl.get_parameter_i32(glow::MAX_ELEMENT_INDEX) } as u32; + + // WORKAROUND: In order to work around an issue with GL on RPI4 and similar, we ignore a + // zero vertex ssbo count if there are vertex sstos. (more info: + // https://github.com/gfx-rs/wgpu/pull/1607#issuecomment-874938961) The hardware does not + // want us to write to these SSBOs, but GLES cannot express that. We detect this case and + // disable writing to SSBOs. + let vertex_ssbo_false_zero = + vertex_shader_storage_blocks == 0 && vertex_shader_storage_textures != 0; + if vertex_ssbo_false_zero { + // We only care about fragment here as the 0 is a lie. + log::debug!("Max vertex shader SSBO == 0 and SSTO != 0. Interpreting as false zero."); + } + + let max_storage_buffers_per_shader_stage = if vertex_shader_storage_blocks == 0 { + fragment_shader_storage_blocks + } else { + vertex_shader_storage_blocks.min(fragment_shader_storage_blocks) + }; + let max_storage_textures_per_shader_stage = if vertex_shader_storage_textures == 0 { + fragment_shader_storage_textures + } else { + vertex_shader_storage_textures.min(fragment_shader_storage_textures) + }; + // NOTE: GL_ARB_compute_shader adds support for indirect dispatch + let indirect_execution = supported((3, 1), (4, 3)) + || (extensions.contains("GL_ARB_draw_indirect") && supports_compute); + let supports_cube_array = supported((3, 2), (4, 0)) + || (supported((3, 1), (4, 0)) && extensions.contains("GL_EXT_texture_cube_map_array")); + + let mut downlevel_flags = wgt::DownlevelFlags::empty() + | wgt::DownlevelFlags::NON_POWER_OF_TWO_MIPMAPPED_TEXTURES + | wgt::DownlevelFlags::COMPARISON_SAMPLERS + | wgt::DownlevelFlags::SHADER_F16_IN_F32; + downlevel_flags.set( + wgt::DownlevelFlags::CUBE_ARRAY_TEXTURES, + supports_cube_array, + ); + downlevel_flags.set(wgt::DownlevelFlags::COMPUTE_SHADERS, supports_compute); + downlevel_flags.set( + wgt::DownlevelFlags::FRAGMENT_WRITABLE_STORAGE, + max_storage_block_size != 0, + ); + downlevel_flags.set(wgt::DownlevelFlags::INDIRECT_EXECUTION, indirect_execution); + downlevel_flags.set(wgt::DownlevelFlags::BASE_VERTEX, supported((3, 2), (3, 2))); + downlevel_flags.set( + wgt::DownlevelFlags::INDEPENDENT_BLEND, + supported((3, 2), (4, 0)) || extensions.contains("GL_EXT_draw_buffers_indexed"), + ); + downlevel_flags.set( + wgt::DownlevelFlags::VERTEX_STORAGE, + max_storage_block_size != 0 + && max_storage_buffers_per_shader_stage != 0 + && (vertex_shader_storage_blocks != 0 || vertex_ssbo_false_zero), + ); + downlevel_flags.set(wgt::DownlevelFlags::FRAGMENT_STORAGE, supports_storage); + if extensions.contains("EXT_texture_filter_anisotropic") + || extensions.contains("GL_EXT_texture_filter_anisotropic") + { + let max_aniso = + unsafe { gl.get_parameter_i32(glow::MAX_TEXTURE_MAX_ANISOTROPY_EXT) } as u32; + downlevel_flags.set(wgt::DownlevelFlags::ANISOTROPIC_FILTERING, max_aniso >= 16); + } + downlevel_flags.set( + wgt::DownlevelFlags::BUFFER_BINDINGS_NOT_16_BYTE_ALIGNED, + !(cfg!(any(webgl, Emscripten)) || is_angle), + ); + // see https://registry.khronos.org/webgl/specs/latest/2.0/#BUFFER_OBJECT_BINDING + downlevel_flags.set( + wgt::DownlevelFlags::UNRESTRICTED_INDEX_BUFFER, + !cfg!(any(webgl, Emscripten)), + ); + downlevel_flags.set( + wgt::DownlevelFlags::UNRESTRICTED_EXTERNAL_TEXTURE_COPIES, + !cfg!(any(webgl, Emscripten)), + ); + downlevel_flags.set( + wgt::DownlevelFlags::FULL_DRAW_INDEX_UINT32, + max_element_index == u32::MAX, + ); + downlevel_flags.set( + wgt::DownlevelFlags::MULTISAMPLED_SHADING, + supported((3, 2), (4, 0)) || extensions.contains("OES_sample_variables"), + ); + let query_buffers = extensions.contains("GL_ARB_query_buffer_object") + || extensions.contains("GL_AMD_query_buffer_object"); + if query_buffers { + downlevel_flags.set(wgt::DownlevelFlags::NONBLOCKING_QUERY_RESOLVE, true); + } + + let mut features = wgt::Features::empty() + | wgt::Features::TEXTURE_ADAPTER_SPECIFIC_FORMAT_FEATURES + | wgt::Features::CLEAR_TEXTURE + | wgt::Features::IMMEDIATES + | wgt::Features::DEPTH32FLOAT_STENCIL8; + features.set( + wgt::Features::ADDRESS_MODE_CLAMP_TO_BORDER | wgt::Features::ADDRESS_MODE_CLAMP_TO_ZERO, + extensions.contains("GL_EXT_texture_border_clamp") + || extensions.contains("GL_ARB_texture_border_clamp"), + ); + features.set( + wgt::Features::DEPTH_CLIP_CONTROL, + extensions.contains("GL_EXT_depth_clamp") || extensions.contains("GL_ARB_depth_clamp"), + ); + features.set( + wgt::Features::VERTEX_WRITABLE_STORAGE, + downlevel_flags.contains(wgt::DownlevelFlags::VERTEX_STORAGE) + && vertex_shader_storage_textures != 0, + ); + features.set( + wgt::Features::MULTIVIEW, + extensions.contains("OVR_multiview2") || extensions.contains("GL_OVR_multiview2"), + ); + features.set( + wgt::Features::DUAL_SOURCE_BLENDING, + extensions.contains("GL_EXT_blend_func_extended") + || extensions.contains("GL_ARB_blend_func_extended"), + ); + features.set( + wgt::Features::CLIP_DISTANCES, + full_ver.is_some() || extensions.contains("GL_EXT_clip_cull_distance"), + ); + features.set( + wgt::Features::PRIMITIVE_INDEX, + supported((3, 2), (3, 2)) + || extensions.contains("OES_geometry_shader") + || extensions.contains("GL_ARB_geometry_shader4"), + ); + features.set( + wgt::Features::SHADER_EARLY_DEPTH_TEST, + supported((3, 1), (4, 2)) || extensions.contains("GL_ARB_shader_image_load_store"), + ); + if extensions.contains("GL_ARB_timer_query") { + features.set(wgt::Features::TIMESTAMP_QUERY, true); + features.set(wgt::Features::TIMESTAMP_QUERY_INSIDE_ENCODERS, true); + features.set(wgt::Features::TIMESTAMP_QUERY_INSIDE_PASSES, true); + } + let gl_bcn_exts = [ + "GL_EXT_texture_compression_s3tc", + "GL_EXT_texture_compression_rgtc", + "GL_ARB_texture_compression_bptc", + ]; + let gles_bcn_exts = [ + "GL_EXT_texture_compression_s3tc_srgb", + "GL_EXT_texture_compression_rgtc", + "GL_EXT_texture_compression_bptc", + ]; + let webgl_bcn_exts = [ + "WEBGL_compressed_texture_s3tc", + "WEBGL_compressed_texture_s3tc_srgb", + "EXT_texture_compression_rgtc", + "EXT_texture_compression_bptc", + ]; + let bcn_exts = if cfg!(any(webgl, Emscripten)) { + &webgl_bcn_exts[..] + } else if es_ver.is_some() { + &gles_bcn_exts[..] + } else { + &gl_bcn_exts[..] + }; + features.set( + wgt::Features::TEXTURE_COMPRESSION_BC, + bcn_exts.iter().all(|&ext| extensions.contains(ext)), + ); + features.set( + wgt::Features::TEXTURE_COMPRESSION_BC_SLICED_3D, + bcn_exts.iter().all(|&ext| extensions.contains(ext)), // BC guaranteed Sliced 3D + ); + let has_etc = if cfg!(any(webgl, Emscripten)) { + extensions.contains("WEBGL_compressed_texture_etc") + } else { + es_ver.is_some() || extensions.contains("GL_ARB_ES3_compatibility") + }; + features.set(wgt::Features::TEXTURE_COMPRESSION_ETC2, has_etc); + + // `OES_texture_compression_astc` provides 2D + 3D, LDR + HDR support + if extensions.contains("WEBGL_compressed_texture_astc") + || extensions.contains("GL_OES_texture_compression_astc") + { + #[cfg(webgl)] + { + if context + .glow_context + .compressed_texture_astc_supports_ldr_profile() + { + features.insert(wgt::Features::TEXTURE_COMPRESSION_ASTC); + features.insert(wgt::Features::TEXTURE_COMPRESSION_ASTC_SLICED_3D); + } + if context + .glow_context + .compressed_texture_astc_supports_hdr_profile() + { + features.insert(wgt::Features::TEXTURE_COMPRESSION_ASTC_HDR); + } + } + + #[cfg(any(native, Emscripten))] + { + features.insert(wgt::Features::TEXTURE_COMPRESSION_ASTC); + features.insert(wgt::Features::TEXTURE_COMPRESSION_ASTC_SLICED_3D); + features.insert(wgt::Features::TEXTURE_COMPRESSION_ASTC_HDR); + } + } else { + features.set( + wgt::Features::TEXTURE_COMPRESSION_ASTC, + extensions.contains("GL_KHR_texture_compression_astc_ldr"), + ); + features.set( + wgt::Features::TEXTURE_COMPRESSION_ASTC_SLICED_3D, + extensions.contains("GL_KHR_texture_compression_astc_ldr") + && extensions.contains("GL_KHR_texture_compression_astc_sliced_3d"), + ); + features.set( + wgt::Features::TEXTURE_COMPRESSION_ASTC_HDR, + extensions.contains("GL_KHR_texture_compression_astc_hdr"), + ); + } + + features.set( + wgt::Features::FLOAT32_FILTERABLE, + extensions.contains("GL_ARB_color_buffer_float") + || extensions.contains("GL_EXT_color_buffer_float") + || extensions.contains("OES_texture_float_linear"), + ); + + if es_ver.is_none() { + features |= wgt::Features::POLYGON_MODE_LINE | wgt::Features::POLYGON_MODE_POINT; + } + + // We *might* be able to emulate bgra8unorm-storage but currently don't attempt to. + + let mut private_caps = super::PrivateCapabilities::empty(); + private_caps.set( + super::PrivateCapabilities::BUFFER_ALLOCATION, + extensions.contains("GL_EXT_buffer_storage") + || extensions.contains("GL_ARB_buffer_storage"), + ); + private_caps.set( + super::PrivateCapabilities::SHADER_BINDING_LAYOUT, + supports_compute, + ); + private_caps.set( + super::PrivateCapabilities::SHADER_TEXTURE_SHADOW_LOD, + extensions.contains("GL_EXT_texture_shadow_lod"), + ); + private_caps.set( + super::PrivateCapabilities::MEMORY_BARRIERS, + supported((3, 1), (4, 2)), + ); + private_caps.set( + super::PrivateCapabilities::VERTEX_BUFFER_LAYOUT, + supported((3, 1), (4, 3)) || extensions.contains("GL_ARB_vertex_attrib_binding"), + ); + private_caps.set( + super::PrivateCapabilities::INDEX_BUFFER_ROLE_CHANGE, + !cfg!(any(webgl, Emscripten)), + ); + private_caps.set( + super::PrivateCapabilities::GET_BUFFER_SUB_DATA, + cfg!(any(webgl, Emscripten)) || full_ver.is_some(), + ); + let color_buffer_float = extensions.contains("GL_EXT_color_buffer_float") + || extensions.contains("GL_ARB_color_buffer_float") + || extensions.contains("EXT_color_buffer_float"); + let color_buffer_half_float = extensions.contains("GL_EXT_color_buffer_half_float") + || extensions.contains("GL_ARB_half_float_pixel"); + private_caps.set( + super::PrivateCapabilities::COLOR_BUFFER_HALF_FLOAT, + color_buffer_half_float || color_buffer_float, + ); + private_caps.set( + super::PrivateCapabilities::COLOR_BUFFER_FLOAT, + color_buffer_float, + ); + private_caps.set(super::PrivateCapabilities::QUERY_BUFFERS, query_buffers); + private_caps.set(super::PrivateCapabilities::QUERY_64BIT, full_ver.is_some()); + private_caps.set( + super::PrivateCapabilities::TEXTURE_STORAGE, + supported((3, 0), (4, 2)), + ); + let is_mali = renderer.to_lowercase().contains("mali"); + let debug_fns_enabled = match backend_options.debug_fns { + wgt::GlDebugFns::Auto => gl.supports_debug() && !is_mali, + wgt::GlDebugFns::ForceEnabled => gl.supports_debug(), + wgt::GlDebugFns::Disabled => false, + }; + private_caps.set(super::PrivateCapabilities::DEBUG_FNS, debug_fns_enabled); + private_caps.set( + super::PrivateCapabilities::INVALIDATE_FRAMEBUFFER, + supported((3, 0), (4, 3)), + ); + if let Some(full_ver) = full_ver { + let supported = + full_ver >= (4, 2) && extensions.contains("GL_ARB_shader_draw_parameters"); + private_caps.set( + super::PrivateCapabilities::FULLY_FEATURED_INSTANCING, + supported, + ); + // Desktop 4.2 and greater specify the first instance parameter. + // + // For all other versions, the behavior is undefined. + // + // We only support indirect first instance when we also have ARB_shader_draw_parameters as + // that's the only way to get gl_InstanceID to work correctly. + features.set(wgt::Features::INDIRECT_FIRST_INSTANCE, supported); + } + private_caps.set( + super::PrivateCapabilities::MULTISAMPLED_RENDER_TO_TEXTURE, + extensions.contains("GL_EXT_multisampled_render_to_texture"), + ); + + // GLSL ES 3.10+ / GLSL 4.30+ natively support coherent/volatile qualifiers + // on storage buffers. These were introduced alongside storage buffer support. + if supports_storage { + features |= wgt::Features::MEMORY_DECORATION_COHERENT + | wgt::Features::MEMORY_DECORATION_VOLATILE; + } + + let max_texture_size = unsafe { gl.get_parameter_i32(glow::MAX_TEXTURE_SIZE) } as u32; + let max_texture_3d_size = unsafe { gl.get_parameter_i32(glow::MAX_3D_TEXTURE_SIZE) } as u32; + + let min_uniform_buffer_offset_alignment = + (unsafe { gl.get_parameter_i32(glow::UNIFORM_BUFFER_OFFSET_ALIGNMENT) } as u32); + let min_storage_buffer_offset_alignment = if supports_storage { + (unsafe { gl.get_parameter_i32(glow::SHADER_STORAGE_BUFFER_OFFSET_ALIGNMENT) } as u32) + } else { + 256 + }; + let max_uniform_buffers_per_shader_stage = + unsafe { gl.get_parameter_i32(glow::MAX_VERTEX_UNIFORM_BLOCKS) } + .min(unsafe { gl.get_parameter_i32(glow::MAX_FRAGMENT_UNIFORM_BLOCKS) }) + as u32; + + let max_compute_workgroups_per_dimension = if supports_work_group_params { + unsafe { gl.get_parameter_indexed_i32(glow::MAX_COMPUTE_WORK_GROUP_COUNT, 0) } + .min(unsafe { gl.get_parameter_indexed_i32(glow::MAX_COMPUTE_WORK_GROUP_COUNT, 1) }) + .min(unsafe { gl.get_parameter_indexed_i32(glow::MAX_COMPUTE_WORK_GROUP_COUNT, 2) }) + as u32 + } else { + 0 + }; + + let max_color_attachments = unsafe { + gl.get_parameter_i32(glow::MAX_COLOR_ATTACHMENTS) + .min(gl.get_parameter_i32(glow::MAX_DRAW_BUFFERS)) as u32 + }; + + // 16 bytes per sample is the maximum size of a color attachment. + let max_color_attachment_bytes_per_sample = + max_color_attachments * wgt::TextureFormat::MAX_TARGET_PIXEL_BYTE_COST; + + let limits = crate::auxil::adjust_raw_limits(wgt::Limits { + max_texture_dimension_1d: max_texture_size, + max_texture_dimension_2d: max_texture_size, + max_texture_dimension_3d: max_texture_3d_size, + max_texture_array_layers: unsafe { + gl.get_parameter_i32(glow::MAX_ARRAY_TEXTURE_LAYERS) + } as u32, + max_bind_groups: crate::MAX_BIND_GROUPS as u32, + // No real limit. + max_bindings_per_bind_group: u32::MAX, + max_dynamic_uniform_buffers_per_pipeline_layout: max_uniform_buffers_per_shader_stage, + max_dynamic_storage_buffers_per_pipeline_layout: max_storage_buffers_per_shader_stage, + max_sampled_textures_per_shader_stage: super::MAX_TEXTURE_SLOTS as u32, + max_samplers_per_shader_stage: super::MAX_SAMPLERS as u32, + max_storage_buffers_per_shader_stage, + max_storage_textures_per_shader_stage, + max_uniform_buffers_per_shader_stage, + max_binding_array_elements_per_shader_stage: 0, + max_binding_array_sampler_elements_per_shader_stage: 0, + max_binding_array_acceleration_structure_elements_per_shader_stage: 0, + max_uniform_buffer_binding_size: unsafe { + gl.get_parameter_i32(glow::MAX_UNIFORM_BLOCK_SIZE) + } as u64, + max_storage_buffer_binding_size: if supports_storage { + unsafe { gl.get_parameter_i32(glow::MAX_SHADER_STORAGE_BLOCK_SIZE) } + } else { + 0 + } as u64, + max_vertex_buffers: if private_caps + .contains(super::PrivateCapabilities::VERTEX_BUFFER_LAYOUT) + { + (unsafe { gl.get_parameter_i32(glow::MAX_VERTEX_ATTRIB_BINDINGS) } as u32) + } else { + 16 // should this be different? + }, + max_vertex_attributes: (unsafe { gl.get_parameter_i32(glow::MAX_VERTEX_ATTRIBS) } + as u32) + .min(super::MAX_VERTEX_ATTRIBUTES as u32), + max_vertex_buffer_array_stride: if private_caps + .contains(super::PrivateCapabilities::VERTEX_BUFFER_LAYOUT) + { + if let Some(full_ver) = full_ver { + if full_ver >= (4, 4) { + // We can query `GL_MAX_VERTEX_ATTRIB_STRIDE` in OpenGL 4.4+ + let value = + (unsafe { gl.get_parameter_i32(glow::MAX_VERTEX_ATTRIB_STRIDE) }) + as u32; + + if value == 0 { + // This should be at least 2048, but the driver for AMD Radeon HD 5870 on + // Windows doesn't recognize `GL_MAX_VERTEX_ATTRIB_STRIDE`. + + log::debug!("Max vertex attribute stride is 0. Assuming it is the OpenGL minimum spec 2048"); + 2048 + } else { + value + } + } else { + log::debug!("Max vertex attribute stride unknown. Assuming it is the OpenGL minimum spec 2048"); + 2048 + } + } else { + (unsafe { gl.get_parameter_i32(glow::MAX_VERTEX_ATTRIB_STRIDE) }) as u32 + } + } else { + !0 + }, + max_immediate_size: super::MAX_IMMEDIATES as u32 * 4, + min_uniform_buffer_offset_alignment, + min_storage_buffer_offset_alignment, + max_inter_stage_shader_variables: { + // MAX_VARYING_COMPONENTS may return 0, because it is deprecated since OpenGL 3.2 core, + // and an OpenGL Context with the core profile and with forward-compatibility=true, + // will make deprecated constants unavailable. + let max_varying_components = + unsafe { gl.get_parameter_i32(glow::MAX_VARYING_COMPONENTS) } as u32; + if max_varying_components == 0 { + // default value for max_inter_stage_shader_variables + 15 + } else { + max_varying_components / 4 + } + }, + max_color_attachments, + max_color_attachment_bytes_per_sample, + max_compute_workgroup_storage_size: if supports_work_group_params { + (unsafe { gl.get_parameter_i32(glow::MAX_COMPUTE_SHARED_MEMORY_SIZE) } as u32) + } else { + 0 + }, + max_compute_invocations_per_workgroup: if supports_work_group_params { + (unsafe { gl.get_parameter_i32(glow::MAX_COMPUTE_WORK_GROUP_INVOCATIONS) } as u32) + } else { + 0 + }, + max_compute_workgroup_size_x: if supports_work_group_params { + (unsafe { gl.get_parameter_indexed_i32(glow::MAX_COMPUTE_WORK_GROUP_SIZE, 0) } + as u32) + } else { + 0 + }, + max_compute_workgroup_size_y: if supports_work_group_params { + (unsafe { gl.get_parameter_indexed_i32(glow::MAX_COMPUTE_WORK_GROUP_SIZE, 1) } + as u32) + } else { + 0 + }, + max_compute_workgroup_size_z: if supports_work_group_params { + (unsafe { gl.get_parameter_indexed_i32(glow::MAX_COMPUTE_WORK_GROUP_SIZE, 2) } + as u32) + } else { + 0 + }, + max_compute_workgroups_per_dimension, + max_buffer_size: i32::MAX as u64, + max_non_sampler_bindings: u32::MAX, + + max_task_mesh_workgroup_total_count: 0, + max_task_mesh_workgroups_per_dimension: 0, + max_task_invocations_per_workgroup: 0, + max_task_invocations_per_dimension: 0, + max_mesh_invocations_per_workgroup: 0, + max_mesh_invocations_per_dimension: 0, + max_task_payload_size: 0, + max_mesh_output_vertices: 0, + max_mesh_output_primitives: 0, + max_mesh_output_layers: 0, + max_mesh_multiview_view_count: 0, + + max_blas_primitive_count: 0, + max_blas_geometry_count: 0, + max_tlas_instance_count: 0, + max_acceleration_structures_per_shader_stage: 0, + + max_multiview_view_count: 0, + }); + + let mut workarounds = super::Workarounds::empty(); + + workarounds.set( + super::Workarounds::EMULATE_BUFFER_MAP, + cfg!(any(webgl, Emscripten)), + ); + + let r = renderer.to_lowercase(); + // Check for Mesa sRGB clear bug. See + // [`super::PrivateCapabilities::MESA_I915_SRGB_SHADER_CLEAR`]. + if context.is_owned() + && r.contains("mesa") + && r.contains("intel") + && r.split(&[' ', '(', ')'][..]) + .any(|substr| substr.len() == 3 && substr.chars().nth(2) == Some('l')) + { + log::debug!( + "Detected skylake derivative running on mesa i915. Clears to srgb textures will \ + use manual shader clears." + ); + workarounds.set(super::Workarounds::MESA_I915_SRGB_SHADER_CLEAR, true); + } + + let downlevel_defaults = wgt::DownlevelLimits {}; + let max_samples = unsafe { gl.get_parameter_i32(glow::MAX_SAMPLES) }; + + // Drop the GL guard so we can move the context into AdapterShared + // ( on Wasm the gl handle is just a ref so we tell clippy to allow + // dropping the ref ) + #[cfg_attr(target_arch = "wasm32", allow(dropping_references))] + drop(gl); + + Some(crate::ExposedAdapter { + adapter: super::Adapter { + shared: Arc::new(super::AdapterShared { + context, + private_caps, + workarounds, + features, + limits: limits.clone(), + options: backend_options, + shading_language_version, + next_shader_id: Default::default(), + program_cache: Default::default(), + es: es_ver.is_some(), + max_msaa_samples: max_samples, + }), + }, + info: Self::make_info(vendor, renderer, version), + features, + capabilities: crate::Capabilities { + limits, + downlevel: wgt::DownlevelCapabilities { + flags: downlevel_flags, + limits: downlevel_defaults, + shader_model: wgt::ShaderModel::Sm5, + }, + alignments: crate::Alignments { + buffer_copy_offset: wgt::BufferSize::new(4).unwrap(), + buffer_copy_pitch: wgt::BufferSize::new(4).unwrap(), + // #6151: `wgpu_hal::gles` doesn't ask Naga to inject bounds + // checks in GLSL, and it doesn't request extensions like + // `KHR_robust_buffer_access_behavior` that would provide + // them, so we can't really implement the checks promised by + // [`crate::BufferBinding`]. + // + // Since this is a pre-existing condition, for the time + // being, provide 1 as the value here, to cause as little + // trouble as possible. + uniform_bounds_check_alignment: wgt::BufferSize::new(1).unwrap(), + raw_tlas_instance_size: 0, + ray_tracing_scratch_buffer_alignment: 0, + }, + cooperative_matrix_properties: Vec::new(), + }, + }) + } + + unsafe fn compile_shader( + source: &str, + gl: &glow::Context, + shader_type: u32, + es: bool, + ) -> Option { + let source = if es { + format!("#version 300 es\nprecision lowp float;\n{source}") + } else { + let version = gl.version(); + if version.major == 3 && version.minor == 0 { + // OpenGL 3.0 only supports this format + format!("#version 130\n{source}") + } else { + // OpenGL 3.1+ support this format + format!("#version 140\n{source}") + } + }; + let shader = unsafe { gl.create_shader(shader_type) }.expect("Could not create shader"); + unsafe { gl.shader_source(shader, &source) }; + unsafe { gl.compile_shader(shader) }; + + if !unsafe { gl.get_shader_compile_status(shader) } { + let msg = unsafe { gl.get_shader_info_log(shader) }; + if !msg.is_empty() { + log::error!("\tShader compile error: {msg}"); + } + unsafe { gl.delete_shader(shader) }; + None + } else { + Some(shader) + } + } + + unsafe fn create_shader_clear_program( + gl: &glow::Context, + es: bool, + ) -> Option { + let program = unsafe { gl.create_program() }.expect("Could not create shader program"); + let vertex = unsafe { + Self::compile_shader( + include_str!("./shaders/clear.vert"), + gl, + glow::VERTEX_SHADER, + es, + )? + }; + let fragment = unsafe { + Self::compile_shader( + include_str!("./shaders/clear.frag"), + gl, + glow::FRAGMENT_SHADER, + es, + )? + }; + unsafe { gl.attach_shader(program, vertex) }; + unsafe { gl.attach_shader(program, fragment) }; + unsafe { gl.link_program(program) }; + + let linked_ok = unsafe { gl.get_program_link_status(program) }; + let msg = unsafe { gl.get_program_info_log(program) }; + if !msg.is_empty() { + log::error!("Shader link error: {msg}"); + } + if !linked_ok { + return None; + } + + let color_uniform_location = unsafe { gl.get_uniform_location(program, "color") } + .expect("Could not find color uniform in shader clear shader"); + unsafe { gl.delete_shader(vertex) }; + unsafe { gl.delete_shader(fragment) }; + + Some(ShaderClearProgram { + program, + color_uniform_location, + }) + } +} + +impl crate::Adapter for super::Adapter { + type A = super::Api; + + unsafe fn open( + &self, + features: wgt::Features, + _limits: &wgt::Limits, + _memory_hints: &wgt::MemoryHints, + ) -> Result, crate::DeviceError> { + let gl = &self.shared.context.lock(); + unsafe { gl.pixel_store_i32(glow::UNPACK_ALIGNMENT, 1) }; + unsafe { gl.pixel_store_i32(glow::PACK_ALIGNMENT, 1) }; + let main_vao = + unsafe { gl.create_vertex_array() }.map_err(|_| crate::DeviceError::OutOfMemory)?; + unsafe { gl.bind_vertex_array(Some(main_vao)) }; + + let zero_buffer = + unsafe { gl.create_buffer() }.map_err(|_| crate::DeviceError::OutOfMemory)?; + unsafe { gl.bind_buffer(glow::COPY_READ_BUFFER, Some(zero_buffer)) }; + let zeroes = vec![0u8; super::ZERO_BUFFER_SIZE]; + unsafe { gl.buffer_data_u8_slice(glow::COPY_READ_BUFFER, &zeroes, glow::STATIC_DRAW) }; + + // Compile the shader program we use for doing manual clears to work around Mesa fastclear + // bug. + + let shader_clear_program = if self + .shared + .workarounds + .contains(super::Workarounds::MESA_I915_SRGB_SHADER_CLEAR) + { + Some(unsafe { + Self::create_shader_clear_program(gl, self.shared.es) + .ok_or(crate::DeviceError::Lost)? + }) + } else { + // If we don't need the workaround, don't waste time and resources compiling the clear program + None + }; + + Ok(crate::OpenDevice { + device: super::Device { + shared: Arc::clone(&self.shared), + main_vao, + #[cfg(all(native, feature = "renderdoc"))] + render_doc: Default::default(), + counters: Default::default(), + }, + queue: super::Queue { + shared: Arc::clone(&self.shared), + features, + draw_fbo: unsafe { gl.create_framebuffer() } + .map_err(|_| crate::DeviceError::OutOfMemory)?, + copy_fbo: unsafe { gl.create_framebuffer() } + .map_err(|_| crate::DeviceError::OutOfMemory)?, + shader_clear_program, + zero_buffer, + temp_query_results: Mutex::new(Vec::new()), + draw_buffer_count: AtomicU8::new(1), + current_index_buffer: Mutex::new(None), + }, + }) + } + + unsafe fn texture_format_capabilities( + &self, + format: wgt::TextureFormat, + ) -> crate::TextureFormatCapabilities { + use crate::TextureFormatCapabilities as Tfc; + use wgt::TextureFormat as Tf; + + let sample_count = { + let max_samples = self.shared.max_msaa_samples; + if max_samples >= 16 { + Tfc::MULTISAMPLE_X2 + | Tfc::MULTISAMPLE_X4 + | Tfc::MULTISAMPLE_X8 + | Tfc::MULTISAMPLE_X16 + } else if max_samples >= 8 { + Tfc::MULTISAMPLE_X2 | Tfc::MULTISAMPLE_X4 | Tfc::MULTISAMPLE_X8 + } else { + // The lowest supported level in GLE3.0/WebGL2 is 4X + // (see GL_MAX_SAMPLES in https://registry.khronos.org/OpenGL-Refpages/es3.0/html/glGet.xhtml). + // On some platforms, like iOS Safari, `get_parameter_i32(MAX_SAMPLES)` returns 0, + // so we always fall back to supporting 4x here. + Tfc::MULTISAMPLE_X2 | Tfc::MULTISAMPLE_X4 + } + }; + + // Base types are pulled from the table in the OpenGLES 3.0 spec in section 3.8. + // + // The storage types are based on table 8.26, in section + // "TEXTURE IMAGE LOADS AND STORES" of OpenGLES-3.2 spec. + let empty = Tfc::empty(); + let base = Tfc::COPY_SRC | Tfc::COPY_DST; + let unfilterable = base | Tfc::SAMPLED; + let depth = base | Tfc::SAMPLED | sample_count | Tfc::DEPTH_STENCIL_ATTACHMENT; + let filterable = unfilterable | Tfc::SAMPLED_LINEAR; + let renderable = + unfilterable | Tfc::COLOR_ATTACHMENT | sample_count | Tfc::MULTISAMPLE_RESOLVE; + let filterable_renderable = filterable | renderable | Tfc::COLOR_ATTACHMENT_BLEND; + let storage = + base | Tfc::STORAGE_READ_WRITE | Tfc::STORAGE_READ_ONLY | Tfc::STORAGE_WRITE_ONLY; + + let feature_fn = |f, caps| { + if self.shared.features.contains(f) { + caps + } else { + empty + } + }; + + let bcn_features = feature_fn(wgt::Features::TEXTURE_COMPRESSION_BC, filterable); + let etc2_features = feature_fn(wgt::Features::TEXTURE_COMPRESSION_ETC2, filterable); + let astc_features = feature_fn(wgt::Features::TEXTURE_COMPRESSION_ASTC, filterable); + let astc_hdr_features = feature_fn(wgt::Features::TEXTURE_COMPRESSION_ASTC_HDR, filterable); + + let private_caps_fn = |f, caps| { + if self.shared.private_caps.contains(f) { + caps + } else { + empty + } + }; + + let half_float_renderable = private_caps_fn( + super::PrivateCapabilities::COLOR_BUFFER_HALF_FLOAT, + Tfc::COLOR_ATTACHMENT + | Tfc::COLOR_ATTACHMENT_BLEND + | sample_count + | Tfc::MULTISAMPLE_RESOLVE, + ); + + let float_renderable = private_caps_fn( + super::PrivateCapabilities::COLOR_BUFFER_FLOAT, + Tfc::COLOR_ATTACHMENT + | Tfc::COLOR_ATTACHMENT_BLEND + | sample_count + | Tfc::MULTISAMPLE_RESOLVE, + ); + + let texture_float_linear = feature_fn(wgt::Features::FLOAT32_FILTERABLE, filterable); + + let image_atomic = feature_fn(wgt::Features::TEXTURE_ATOMIC, Tfc::STORAGE_ATOMIC); + let image_64_atomic = feature_fn(wgt::Features::TEXTURE_INT64_ATOMIC, Tfc::STORAGE_ATOMIC); + + match format { + Tf::R8Unorm => filterable_renderable, + Tf::R8Snorm => filterable, + Tf::R8Uint => renderable, + Tf::R8Sint => renderable, + Tf::R16Uint => renderable, + Tf::R16Sint => renderable, + Tf::R16Unorm => empty, + Tf::R16Snorm => empty, + Tf::R16Float => filterable | half_float_renderable, + Tf::Rg8Unorm => filterable_renderable, + Tf::Rg8Snorm => filterable, + Tf::Rg8Uint => renderable, + Tf::Rg8Sint => renderable, + Tf::R32Uint => renderable | storage | image_atomic, + Tf::R32Sint => renderable | storage | image_atomic, + Tf::R32Float => unfilterable | storage | float_renderable | texture_float_linear, + Tf::Rg16Uint => renderable, + Tf::Rg16Sint => renderable, + Tf::Rg16Unorm => empty, + Tf::Rg16Snorm => empty, + Tf::Rg16Float => filterable | half_float_renderable, + Tf::Rgba8Unorm => filterable_renderable | storage, + Tf::Rgba8UnormSrgb => filterable_renderable, + Tf::Bgra8Unorm | Tf::Bgra8UnormSrgb => filterable_renderable, + Tf::Rgba8Snorm => filterable | storage, + Tf::Rgba8Uint => renderable | storage, + Tf::Rgba8Sint => renderable | storage, + Tf::Rgb10a2Uint => renderable, + Tf::Rgb10a2Unorm => filterable_renderable, + Tf::Rg11b10Ufloat => filterable | float_renderable, + Tf::R64Uint => image_64_atomic, + Tf::Rg32Uint => renderable, + Tf::Rg32Sint => renderable, + Tf::Rg32Float => unfilterable | float_renderable | texture_float_linear, + Tf::Rgba16Uint => renderable | storage, + Tf::Rgba16Sint => renderable | storage, + Tf::Rgba16Unorm => empty, + Tf::Rgba16Snorm => empty, + Tf::Rgba16Float => filterable | storage | half_float_renderable, + Tf::Rgba32Uint => renderable | storage, + Tf::Rgba32Sint => renderable | storage, + Tf::Rgba32Float => unfilterable | storage | float_renderable | texture_float_linear, + Tf::Stencil8 + | Tf::Depth16Unorm + | Tf::Depth32Float + | Tf::Depth32FloatStencil8 + | Tf::Depth24Plus + | Tf::Depth24PlusStencil8 => depth, + Tf::NV12 => empty, + Tf::P010 => empty, + Tf::Rgb9e5Ufloat => filterable, + Tf::Bc1RgbaUnorm + | Tf::Bc1RgbaUnormSrgb + | Tf::Bc2RgbaUnorm + | Tf::Bc2RgbaUnormSrgb + | Tf::Bc3RgbaUnorm + | Tf::Bc3RgbaUnormSrgb + | Tf::Bc4RUnorm + | Tf::Bc4RSnorm + | Tf::Bc5RgUnorm + | Tf::Bc5RgSnorm + | Tf::Bc6hRgbFloat + | Tf::Bc6hRgbUfloat + | Tf::Bc7RgbaUnorm + | Tf::Bc7RgbaUnormSrgb => bcn_features, + Tf::Etc2Rgb8Unorm + | Tf::Etc2Rgb8UnormSrgb + | Tf::Etc2Rgb8A1Unorm + | Tf::Etc2Rgb8A1UnormSrgb + | Tf::Etc2Rgba8Unorm + | Tf::Etc2Rgba8UnormSrgb + | Tf::EacR11Unorm + | Tf::EacR11Snorm + | Tf::EacRg11Unorm + | Tf::EacRg11Snorm => etc2_features, + Tf::Astc { + block: _, + channel: AstcChannel::Unorm | AstcChannel::UnormSrgb, + } => astc_features, + Tf::Astc { + block: _, + channel: AstcChannel::Hdr, + } => astc_hdr_features, + } + } + + unsafe fn surface_capabilities( + &self, + surface: &super::Surface, + ) -> Option { + #[cfg(webgl)] + if self.shared.context.webgl2_context != surface.webgl2_context { + return None; + } + + if surface.presentable { + let mut formats = vec![ + wgt::TextureFormat::Rgba8Unorm, + #[cfg(native)] + wgt::TextureFormat::Bgra8Unorm, + ]; + if surface.supports_srgb() { + formats.extend([ + wgt::TextureFormat::Rgba8UnormSrgb, + #[cfg(native)] + wgt::TextureFormat::Bgra8UnormSrgb, + ]) + } + if self + .shared + .private_caps + .contains(super::PrivateCapabilities::COLOR_BUFFER_HALF_FLOAT) + { + formats.push(wgt::TextureFormat::Rgba16Float) + } + + Some(crate::SurfaceCapabilities { + formats, + present_modes: if cfg!(windows) { + vec![wgt::PresentMode::Fifo, wgt::PresentMode::Immediate] + } else { + vec![wgt::PresentMode::Fifo] //TODO + }, + composite_alpha_modes: vec![wgt::CompositeAlphaMode::Opaque], //TODO + maximum_frame_latency: 2..=2, //TODO, unused currently + current_extent: None, + usage: wgt::TextureUses::COLOR_TARGET, + }) + } else { + None + } + } + + unsafe fn get_presentation_timestamp(&self) -> wgt::PresentationTimestamp { + wgt::PresentationTimestamp::INVALID_TIMESTAMP + } + + fn get_ordered_buffer_usages(&self) -> wgt::BufferUses { + wgt::BufferUses::INCLUSIVE | wgt::BufferUses::MAP_WRITE + } + + // Don't put barriers between inclusive uses + fn get_ordered_texture_usages(&self) -> wgt::TextureUses { + wgt::TextureUses::INCLUSIVE + | wgt::TextureUses::COLOR_TARGET + | wgt::TextureUses::DEPTH_STENCIL_WRITE + } +} + +impl super::AdapterShared { + pub(super) unsafe fn get_buffer_sub_data( + &self, + gl: &glow::Context, + target: u32, + offset: i32, + dst_data: &mut [u8], + ) { + if self + .private_caps + .contains(super::PrivateCapabilities::GET_BUFFER_SUB_DATA) + { + unsafe { gl.get_buffer_sub_data(target, offset, dst_data) }; + } else { + log::error!("Fake map"); + let length = dst_data.len(); + let buffer_mapping = + unsafe { gl.map_buffer_range(target, offset, length as _, glow::MAP_READ_BIT) }; + + unsafe { + core::ptr::copy_nonoverlapping(buffer_mapping, dst_data.as_mut_ptr(), length) + }; + + unsafe { gl.unmap_buffer(target) }; + } + } +} + +#[cfg(send_sync)] +unsafe impl Sync for super::Adapter {} +#[cfg(send_sync)] +unsafe impl Send for super::Adapter {} + +#[cfg(test)] +mod tests { + use super::super::Adapter; + + #[test] + fn test_version_parse() { + Adapter::parse_version("1").unwrap_err(); + Adapter::parse_version("1.").unwrap_err(); + Adapter::parse_version("1 h3l1o. W0rld").unwrap_err(); + Adapter::parse_version("1. h3l1o. W0rld").unwrap_err(); + Adapter::parse_version("1.2.3").unwrap_err(); + + assert_eq!(Adapter::parse_version("OpenGL ES 3.1").unwrap(), (3, 1)); + assert_eq!( + Adapter::parse_version("OpenGL ES 2.0 Google Nexus").unwrap(), + (2, 0) + ); + assert_eq!(Adapter::parse_version("GLSL ES 1.1").unwrap(), (1, 1)); + assert_eq!( + Adapter::parse_version("OpenGL ES GLSL ES 3.20").unwrap(), + (3, 2) + ); + assert_eq!( + // WebGL 2.0 should parse as OpenGL ES 3.0 + Adapter::parse_version("WebGL 2.0 (OpenGL ES 3.0 Chromium)").unwrap(), + (3, 0) + ); + assert_eq!( + Adapter::parse_version("WebGL GLSL ES 3.00 (OpenGL ES GLSL ES 3.0 Chromium)").unwrap(), + (3, 0) + ); + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/gles/command.rs b/third_party/wgpu-hal-29.0.4/src/gles/command.rs new file mode 100644 index 0000000..b03a560 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/command.rs @@ -0,0 +1,1297 @@ +use alloc::string::String; +use core::{mem, ops::Range}; + +use arrayvec::ArrayVec; + +use super::{conv, Command as C}; + +#[derive(Clone, Copy, Debug, Default)] +struct TextureSlotDesc { + tex_target: super::BindTarget, + sampler_index: Option, +} + +pub(super) struct State { + topology: u32, + primitive: super::PrimitiveState, + index_format: wgt::IndexFormat, + index_offset: wgt::BufferAddress, + vertex_buffers: + [(super::VertexBufferDesc, Option); crate::MAX_VERTEX_BUFFERS], + vertex_attributes: ArrayVec, + color_targets: ArrayVec, + stencil: super::StencilState, + depth_bias: wgt::DepthBiasState, + alpha_to_coverage_enabled: bool, + samplers: [Option; super::MAX_SAMPLERS], + texture_slots: [TextureSlotDesc; super::MAX_TEXTURE_SLOTS], + render_size: wgt::Extent3d, + resolve_attachments: ArrayVec<(u32, super::TextureView), { crate::MAX_COLOR_ATTACHMENTS }>, + invalidate_attachments: ArrayVec, + has_pass_label: bool, + instance_vbuf_mask: usize, + dirty_vbuf_mask: usize, + active_first_instance: u32, + first_instance_location: Option, + immediates_descs: ArrayVec, + // The current state of the immediate data block. + current_immediates_data: [u32; super::MAX_IMMEDIATES], + end_of_pass_timestamp: Option, + clip_distance_count: u32, +} + +impl Default for State { + fn default() -> Self { + Self { + topology: Default::default(), + primitive: Default::default(), + index_format: Default::default(), + index_offset: Default::default(), + vertex_buffers: Default::default(), + vertex_attributes: Default::default(), + color_targets: Default::default(), + stencil: Default::default(), + depth_bias: Default::default(), + alpha_to_coverage_enabled: Default::default(), + samplers: Default::default(), + texture_slots: Default::default(), + render_size: Default::default(), + resolve_attachments: Default::default(), + invalidate_attachments: Default::default(), + has_pass_label: Default::default(), + instance_vbuf_mask: Default::default(), + dirty_vbuf_mask: Default::default(), + active_first_instance: Default::default(), + first_instance_location: Default::default(), + immediates_descs: Default::default(), + current_immediates_data: [0; super::MAX_IMMEDIATES], + end_of_pass_timestamp: Default::default(), + clip_distance_count: Default::default(), + } + } +} + +impl super::CommandBuffer { + fn clear(&mut self) { + self.label = None; + self.commands.clear(); + self.data_bytes.clear(); + self.queries.clear(); + } + + fn add_marker(&mut self, marker: &str) -> Range { + let start = self.data_bytes.len() as u32; + self.data_bytes.extend(marker.as_bytes()); + start..self.data_bytes.len() as u32 + } + + fn add_immediates_data(&mut self, data: &[u32]) -> Range { + let data_raw = bytemuck::cast_slice(data); + let start = self.data_bytes.len(); + assert!(start < u32::MAX as usize); + self.data_bytes.extend_from_slice(data_raw); + let end = self.data_bytes.len(); + assert!(end < u32::MAX as usize); + (start as u32)..(end as u32) + } +} + +impl Drop for super::CommandEncoder { + fn drop(&mut self) { + use crate::CommandEncoder; + unsafe { self.discard_encoding() } + self.counters.command_encoders.sub(1); + } +} + +impl super::CommandEncoder { + fn rebind_stencil_func(&mut self) { + fn make(s: &super::StencilSide, face: u32) -> C { + C::SetStencilFunc { + face, + function: s.function, + reference: s.reference, + read_mask: s.mask_read, + } + } + + let s = &self.state.stencil; + if s.front.function == s.back.function + && s.front.mask_read == s.back.mask_read + && s.front.reference == s.back.reference + { + self.cmd_buffer + .commands + .push(make(&s.front, glow::FRONT_AND_BACK)); + } else { + self.cmd_buffer.commands.push(make(&s.front, glow::FRONT)); + self.cmd_buffer.commands.push(make(&s.back, glow::BACK)); + } + } + + fn rebind_vertex_data(&mut self, first_instance: u32) { + if self + .private_caps + .contains(super::PrivateCapabilities::VERTEX_BUFFER_LAYOUT) + { + for (index, pair) in self.state.vertex_buffers.iter().enumerate() { + if self.state.dirty_vbuf_mask & (1 << index) == 0 { + continue; + } + let (buffer_desc, vb) = match *pair { + // Not all dirty bindings are necessarily filled. Some may be unused. + (_, None) => continue, + (ref vb_desc, Some(ref vb)) => (vb_desc.clone(), vb), + }; + let instance_offset = match buffer_desc.step { + wgt::VertexStepMode::Vertex => 0, + wgt::VertexStepMode::Instance => first_instance * buffer_desc.stride, + }; + + self.cmd_buffer.commands.push(C::SetVertexBuffer { + index: index as u32, + buffer: super::BufferBinding { + raw: vb.raw, + offset: vb.offset + instance_offset as wgt::BufferAddress, + }, + buffer_desc, + }); + self.state.dirty_vbuf_mask ^= 1 << index; + } + } else { + let mut vbuf_mask = 0; + for attribute in self.state.vertex_attributes.iter() { + if self.state.dirty_vbuf_mask & (1 << attribute.buffer_index) == 0 { + continue; + } + let (buffer_desc, vb) = + match self.state.vertex_buffers[attribute.buffer_index as usize] { + // Not all dirty bindings are necessarily filled. Some may be unused. + (_, None) => continue, + (ref vb_desc, Some(ref vb)) => (vb_desc.clone(), vb), + }; + + let mut attribute_desc = attribute.clone(); + attribute_desc.offset += vb.offset as u32; + if buffer_desc.step == wgt::VertexStepMode::Instance { + attribute_desc.offset += buffer_desc.stride * first_instance; + } + + self.cmd_buffer.commands.push(C::SetVertexAttribute { + buffer: Some(vb.raw), + buffer_desc, + attribute_desc, + }); + vbuf_mask |= 1 << attribute.buffer_index; + } + self.state.dirty_vbuf_mask ^= vbuf_mask; + } + } + + fn rebind_sampler_states(&mut self, dirty_textures: u32, dirty_samplers: u32) { + for (texture_index, slot) in self.state.texture_slots.iter().enumerate() { + if dirty_textures & (1 << texture_index) != 0 + || slot + .sampler_index + .is_some_and(|si| dirty_samplers & (1 << si) != 0) + { + let sampler = slot + .sampler_index + .and_then(|si| self.state.samplers[si as usize]); + self.cmd_buffer + .commands + .push(C::BindSampler(texture_index as u32, sampler)); + } + } + } + + fn prepare_draw(&mut self, first_instance: u32) { + // If we support fully featured instancing, we want to bind everything as normal + // and let the draw call sort it out. + let emulated_first_instance_value = if self + .private_caps + .contains(super::PrivateCapabilities::FULLY_FEATURED_INSTANCING) + { + 0 + } else { + first_instance + }; + + if emulated_first_instance_value != self.state.active_first_instance { + // rebind all per-instance buffers on first-instance change + self.state.dirty_vbuf_mask |= self.state.instance_vbuf_mask; + self.state.active_first_instance = emulated_first_instance_value; + } + if self.state.dirty_vbuf_mask != 0 { + self.rebind_vertex_data(emulated_first_instance_value); + } + } + + fn set_pipeline_inner(&mut self, inner: &super::PipelineInner) { + self.cmd_buffer.commands.push(C::SetProgram(inner.program)); + + self.state + .first_instance_location + .clone_from(&inner.first_instance_location); + self.state + .immediates_descs + .clone_from(&inner.immediates_descs); + + // rebind textures, if needed + let mut dirty_textures = 0u32; + for (texture_index, (slot, &sampler_index)) in self + .state + .texture_slots + .iter_mut() + .zip(inner.sampler_map.iter()) + .enumerate() + { + if slot.sampler_index != sampler_index { + slot.sampler_index = sampler_index; + dirty_textures |= 1 << texture_index; + } + } + if dirty_textures != 0 { + self.rebind_sampler_states(dirty_textures, 0); + } + } +} + +impl crate::CommandEncoder for super::CommandEncoder { + type A = super::Api; + + unsafe fn begin_encoding(&mut self, label: crate::Label) -> Result<(), crate::DeviceError> { + self.state = State::default(); + self.cmd_buffer.label = label.map(String::from); + Ok(()) + } + unsafe fn discard_encoding(&mut self) { + self.cmd_buffer.clear(); + } + unsafe fn end_encoding(&mut self) -> Result { + Ok(mem::take(&mut self.cmd_buffer)) + } + unsafe fn reset_all(&mut self, _command_buffers: I) { + //TODO: could re-use the allocations in all these command buffers + } + + unsafe fn transition_buffers<'a, T>(&mut self, barriers: T) + where + T: Iterator>, + { + if !self + .private_caps + .contains(super::PrivateCapabilities::MEMORY_BARRIERS) + { + return; + } + for bar in barriers { + // GLES only synchronizes storage -> anything explicitly + if !bar.usage.from.contains(wgt::BufferUses::STORAGE_READ_WRITE) { + continue; + } + self.cmd_buffer + .commands + .push(C::BufferBarrier(bar.buffer.raw.unwrap(), bar.usage.to)); + } + } + + unsafe fn transition_textures<'a, T>(&mut self, barriers: T) + where + T: Iterator>, + { + if !self + .private_caps + .contains(super::PrivateCapabilities::MEMORY_BARRIERS) + { + return; + } + + let mut combined_usage = wgt::TextureUses::empty(); + for bar in barriers { + // GLES only synchronizes storage -> anything explicitly + // if shader writes to a texture then barriers should be placed + if !bar.usage.from.intersects( + wgt::TextureUses::STORAGE_READ_WRITE | wgt::TextureUses::STORAGE_WRITE_ONLY, + ) { + continue; + } + // unlike buffers, there is no need for a concrete texture + // object to be bound anywhere for a barrier + combined_usage |= bar.usage.to; + } + + if !combined_usage.is_empty() { + self.cmd_buffer + .commands + .push(C::TextureBarrier(combined_usage)); + } + } + + unsafe fn clear_buffer(&mut self, buffer: &super::Buffer, range: crate::MemoryRange) { + self.cmd_buffer.commands.push(C::ClearBuffer { + dst: buffer.clone(), + dst_target: buffer.target, + range, + }); + } + + unsafe fn copy_buffer_to_buffer( + &mut self, + src: &super::Buffer, + dst: &super::Buffer, + regions: T, + ) where + T: Iterator, + { + let (src_target, dst_target) = if src.target == dst.target { + (glow::COPY_READ_BUFFER, glow::COPY_WRITE_BUFFER) + } else { + (src.target, dst.target) + }; + for copy in regions { + self.cmd_buffer.commands.push(C::CopyBufferToBuffer { + src: src.clone(), + src_target, + dst: dst.clone(), + dst_target, + copy, + }) + } + } + + #[cfg(webgl)] + unsafe fn copy_external_image_to_texture( + &mut self, + src: &wgt::CopyExternalImageSourceInfo, + dst: &super::Texture, + dst_premultiplication: bool, + regions: T, + ) where + T: Iterator, + { + let (dst_raw, dst_target) = dst.inner.as_native(); + for copy in regions { + self.cmd_buffer + .commands + .push(C::CopyExternalImageToTexture { + src: src.clone(), + dst: dst_raw, + dst_target, + dst_format: dst.format, + dst_premultiplication, + copy, + }) + } + } + + unsafe fn copy_texture_to_texture( + &mut self, + src: &super::Texture, + _src_usage: wgt::TextureUses, + dst: &super::Texture, + regions: T, + ) where + T: Iterator, + { + let (src_raw, src_target) = src.inner.as_native(); + let (dst_raw, dst_target) = dst.inner.as_native(); + for mut copy in regions { + copy.clamp_size_to_virtual(&src.copy_size, &dst.copy_size); + self.cmd_buffer.commands.push(C::CopyTextureToTexture { + src: src_raw, + src_target, + dst: dst_raw, + dst_target, + copy, + }) + } + } + + unsafe fn copy_buffer_to_texture( + &mut self, + src: &super::Buffer, + dst: &super::Texture, + regions: T, + ) where + T: Iterator, + { + let (dst_raw, dst_target) = dst.inner.as_native(); + + for mut copy in regions { + copy.clamp_size_to_virtual(&dst.copy_size); + self.cmd_buffer.commands.push(C::CopyBufferToTexture { + src: src.clone(), + src_target: src.target, + dst: dst_raw, + dst_target, + dst_format: dst.format, + copy, + }) + } + } + + unsafe fn copy_texture_to_buffer( + &mut self, + src: &super::Texture, + _src_usage: wgt::TextureUses, + dst: &super::Buffer, + regions: T, + ) where + T: Iterator, + { + let (src_raw, src_target) = src.inner.as_native(); + for mut copy in regions { + copy.clamp_size_to_virtual(&src.copy_size); + self.cmd_buffer.commands.push(C::CopyTextureToBuffer { + src: src_raw, + src_target, + src_format: src.format, + dst: dst.clone(), + dst_target: dst.target, + copy, + }) + } + } + + unsafe fn begin_query(&mut self, set: &super::QuerySet, index: u32) { + let query = set.queries[index as usize]; + self.cmd_buffer + .commands + .push(C::BeginQuery(query, set.target)); + } + unsafe fn end_query(&mut self, set: &super::QuerySet, _index: u32) { + self.cmd_buffer.commands.push(C::EndQuery(set.target)); + } + unsafe fn write_timestamp(&mut self, set: &super::QuerySet, index: u32) { + let query = set.queries[index as usize]; + self.cmd_buffer.commands.push(C::TimestampQuery(query)); + } + unsafe fn reset_queries(&mut self, _set: &super::QuerySet, _range: Range) { + //TODO: what do we do here? + } + unsafe fn copy_query_results( + &mut self, + set: &super::QuerySet, + range: Range, + buffer: &super::Buffer, + offset: wgt::BufferAddress, + _stride: wgt::BufferSize, + ) { + let start = self.cmd_buffer.queries.len(); + self.cmd_buffer + .queries + .extend_from_slice(&set.queries[range.start as usize..range.end as usize]); + let query_range = start as u32..self.cmd_buffer.queries.len() as u32; + self.cmd_buffer.commands.push(C::CopyQueryResults { + query_range, + dst: buffer.clone(), + dst_target: buffer.target, + dst_offset: offset, + }); + } + + // render + + unsafe fn begin_render_pass( + &mut self, + desc: &crate::RenderPassDescriptor, + ) -> Result<(), crate::DeviceError> { + debug_assert!(self.state.end_of_pass_timestamp.is_none()); + if let Some(ref t) = desc.timestamp_writes { + if let Some(index) = t.beginning_of_pass_write_index { + unsafe { self.write_timestamp(t.query_set, index) } + } + self.state.end_of_pass_timestamp = t + .end_of_pass_write_index + .map(|index| t.query_set.queries[index as usize]); + } + + self.state.render_size = desc.extent; + self.state.resolve_attachments.clear(); + self.state.invalidate_attachments.clear(); + if let Some(label) = desc.label { + let range = self.cmd_buffer.add_marker(label); + self.cmd_buffer.commands.push(C::PushDebugGroup(range)); + self.state.has_pass_label = true; + } + + let rendering_to_external_framebuffer = desc + .color_attachments + .iter() + .filter_map(|at| at.as_ref()) + .any(|at| match at.target.view.inner { + #[cfg(webgl)] + super::TextureInner::ExternalFramebuffer { .. } => true, + #[cfg(native)] + super::TextureInner::ExternalNativeFramebuffer { .. } => true, + _ => false, + }); + + if rendering_to_external_framebuffer && desc.color_attachments.len() != 1 { + panic!("Multiple render attachments with external framebuffers are not supported."); + } + + // `COLOR_ATTACHMENT0` to `COLOR_ATTACHMENT31` gives 32 possible color attachments. + assert!(desc.color_attachments.len() <= 32); + + match desc + .color_attachments + .first() + .filter(|at| at.is_some()) + .and_then(|at| at.as_ref().map(|at| &at.target.view.inner)) + { + // default framebuffer (provided externally) + Some(&super::TextureInner::DefaultRenderbuffer) => { + self.cmd_buffer + .commands + .push(C::ResetFramebuffer { is_default: true }); + } + _ => { + // set the framebuffer + self.cmd_buffer + .commands + .push(C::ResetFramebuffer { is_default: false }); + + for (i, cat) in desc.color_attachments.iter().enumerate() { + if let Some(cat) = cat.as_ref() { + let attachment = glow::COLOR_ATTACHMENT0 + i as u32; + // Try to use the multisampled render-to-texture extension to avoid resolving + if let Some(ref rat) = cat.resolve_target { + if matches!(rat.view.inner, super::TextureInner::Texture { .. }) + && self.private_caps.contains( + super::PrivateCapabilities::MULTISAMPLED_RENDER_TO_TEXTURE, + ) + && !cat.ops.contains(crate::AttachmentOps::STORE) + // Extension specifies that only COLOR_ATTACHMENT0 is valid + && i == 0 + { + self.cmd_buffer.commands.push(C::BindAttachment { + attachment, + view: rat.view.clone(), + depth_slice: None, + sample_count: desc.sample_count, + }); + continue; + } + } + self.cmd_buffer.commands.push(C::BindAttachment { + attachment, + view: cat.target.view.clone(), + depth_slice: cat.depth_slice, + sample_count: 1, + }); + if let Some(ref rat) = cat.resolve_target { + self.state + .resolve_attachments + .push((attachment, rat.view.clone())); + } + if cat.ops.contains(crate::AttachmentOps::STORE_DISCARD) { + self.state.invalidate_attachments.push(attachment); + } + } + } + if let Some(ref dsat) = desc.depth_stencil_attachment { + let aspects = dsat.target.view.aspects; + let attachment = match aspects { + crate::FormatAspects::DEPTH => glow::DEPTH_ATTACHMENT, + crate::FormatAspects::STENCIL => glow::STENCIL_ATTACHMENT, + _ => glow::DEPTH_STENCIL_ATTACHMENT, + }; + self.cmd_buffer.commands.push(C::BindAttachment { + attachment, + view: dsat.target.view.clone(), + depth_slice: None, + sample_count: 1, + }); + if aspects.contains(crate::FormatAspects::DEPTH) + && dsat.depth_ops.contains(crate::AttachmentOps::STORE_DISCARD) + { + self.state + .invalidate_attachments + .push(glow::DEPTH_ATTACHMENT); + } + if aspects.contains(crate::FormatAspects::STENCIL) + && dsat + .stencil_ops + .contains(crate::AttachmentOps::STORE_DISCARD) + { + self.state + .invalidate_attachments + .push(glow::STENCIL_ATTACHMENT); + } + } + } + } + + let rect = crate::Rect { + x: 0, + y: 0, + w: desc.extent.width as i32, + h: desc.extent.height as i32, + }; + self.cmd_buffer.commands.push(C::SetScissor(rect.clone())); + self.cmd_buffer.commands.push(C::SetViewport { + rect, + depth: 0.0..1.0, + }); + + if !rendering_to_external_framebuffer { + // set the draw buffers and states + self.cmd_buffer + .commands + .push(C::SetDrawColorBuffers(desc.color_attachments.len() as u8)); + } + + // issue the clears + for (i, cat) in desc + .color_attachments + .iter() + .filter_map(|at| at.as_ref()) + .enumerate() + { + if cat.ops.contains(crate::AttachmentOps::LOAD_CLEAR) { + let c = &cat.clear_value; + self.cmd_buffer.commands.push( + match cat.target.view.format.sample_type(None, None).unwrap() { + wgt::TextureSampleType::Float { .. } => C::ClearColorF { + draw_buffer: i as u32, + color: [c.r as f32, c.g as f32, c.b as f32, c.a as f32], + is_srgb: cat.target.view.format.is_srgb(), + }, + wgt::TextureSampleType::Uint => C::ClearColorU( + i as u32, + [c.r as u32, c.g as u32, c.b as u32, c.a as u32], + ), + wgt::TextureSampleType::Sint => C::ClearColorI( + i as u32, + [c.r as i32, c.g as i32, c.b as i32, c.a as i32], + ), + wgt::TextureSampleType::Depth => unreachable!(), + }, + ); + } + } + + if let Some(ref dsat) = desc.depth_stencil_attachment { + let clear_depth = dsat.depth_ops.contains(crate::AttachmentOps::LOAD_CLEAR); + let clear_stencil = dsat.stencil_ops.contains(crate::AttachmentOps::LOAD_CLEAR); + + if clear_depth && clear_stencil { + self.cmd_buffer.commands.push(C::ClearDepthAndStencil( + dsat.clear_value.0, + dsat.clear_value.1, + )); + } else if clear_depth { + self.cmd_buffer + .commands + .push(C::ClearDepth(dsat.clear_value.0)); + } else if clear_stencil { + self.cmd_buffer + .commands + .push(C::ClearStencil(dsat.clear_value.1)); + } + } + Ok(()) + } + unsafe fn end_render_pass(&mut self) { + for (attachment, dst) in self.state.resolve_attachments.drain(..) { + self.cmd_buffer.commands.push(C::ResolveAttachment { + attachment, + dst, + size: self.state.render_size, + }); + } + if !self.state.invalidate_attachments.is_empty() { + self.cmd_buffer.commands.push(C::InvalidateAttachments( + self.state.invalidate_attachments.clone(), + )); + self.state.invalidate_attachments.clear(); + } + if self.state.has_pass_label { + self.cmd_buffer.commands.push(C::PopDebugGroup); + self.state.has_pass_label = false; + } + self.state.instance_vbuf_mask = 0; + self.state.dirty_vbuf_mask = 0; + self.state.active_first_instance = 0; + self.state.color_targets.clear(); + for vat in &self.state.vertex_attributes { + self.cmd_buffer + .commands + .push(C::UnsetVertexAttribute(vat.location)); + } + self.state.vertex_attributes.clear(); + self.state.primitive = super::PrimitiveState::default(); + + if let Some(query) = self.state.end_of_pass_timestamp.take() { + self.cmd_buffer.commands.push(C::TimestampQuery(query)); + } + } + + unsafe fn set_bind_group( + &mut self, + layout: &super::PipelineLayout, + index: u32, + group: &super::BindGroup, + dynamic_offsets: &[wgt::DynamicOffset], + ) { + let mut do_index = 0; + let mut dirty_textures = 0u32; + let mut dirty_samplers = 0u32; + let group_info = layout.group_infos[index as usize].as_ref().unwrap(); + + for (binding_layout, raw_binding) in group_info.entries.iter().zip(group.contents.iter()) { + let slot = group_info.binding_to_slot[binding_layout.binding as usize] as u32; + match *raw_binding { + super::RawBinding::Buffer { + raw, + offset: base_offset, + size, + } => { + let mut offset = base_offset; + let target = match binding_layout.ty { + wgt::BindingType::Buffer { + ty, + has_dynamic_offset, + min_binding_size: _, + } => { + if has_dynamic_offset { + offset += dynamic_offsets[do_index] as i32; + do_index += 1; + } + match ty { + wgt::BufferBindingType::Uniform => glow::UNIFORM_BUFFER, + wgt::BufferBindingType::Storage { .. } => { + glow::SHADER_STORAGE_BUFFER + } + } + } + _ => unreachable!(), + }; + self.cmd_buffer.commands.push(C::BindBuffer { + target, + slot, + buffer: raw, + offset, + size, + }); + } + super::RawBinding::Sampler(sampler) => { + dirty_samplers |= 1 << slot; + self.state.samplers[slot as usize] = Some(sampler); + } + super::RawBinding::Texture { + raw, + target, + aspects, + ref mip_levels, + } => { + dirty_textures |= 1 << slot; + self.state.texture_slots[slot as usize].tex_target = target; + self.cmd_buffer.commands.push(C::BindTexture { + slot, + texture: raw, + target, + aspects, + mip_levels: mip_levels.clone(), + }); + } + super::RawBinding::Image(ref binding) => { + self.cmd_buffer.commands.push(C::BindImage { + slot, + binding: binding.clone(), + }); + } + } + } + + self.rebind_sampler_states(dirty_textures, dirty_samplers); + } + + unsafe fn set_immediates( + &mut self, + _layout: &super::PipelineLayout, + offset_bytes: u32, + data: &[u32], + ) { + // There is nothing preventing the user from trying to update a single value within + // a vector or matrix in the set_immediates call, as to the user, all of this is + // just memory. However OpenGL does not allow partial uniform updates. + // + // As such, we locally keep a copy of the current state of the immediate data memory + // block. If the user tries to update a single value, we have the data to update the entirety + // of the uniform. + let start_words = offset_bytes / 4; + let end_words = start_words + data.len() as u32; + self.state.current_immediates_data[start_words as usize..end_words as usize] + .copy_from_slice(data); + + // We iterate over the uniform list as there may be multiple uniforms that need + // updating from the same immediate data memory (one for each shader stage). + // + // Additionally, any statically unused uniform descs will have been removed from this list + // by OpenGL, so the uniform list is not contiguous. + for uniform in self.state.immediates_descs.iter().cloned() { + let uniform_size_words = uniform.size_bytes / 4; + let uniform_start_words = uniform.offset / 4; + let uniform_end_words = uniform_start_words + uniform_size_words; + + // Is true if any word within the uniform binding was updated + let needs_updating = + start_words < uniform_end_words || uniform_start_words <= end_words; + + if needs_updating { + let uniform_data = &self.state.current_immediates_data + [uniform_start_words as usize..uniform_end_words as usize]; + + let range = self.cmd_buffer.add_immediates_data(uniform_data); + + self.cmd_buffer.commands.push(C::SetImmediates { + uniform, + offset: range.start, + }); + } + } + } + + unsafe fn insert_debug_marker(&mut self, label: &str) { + let range = self.cmd_buffer.add_marker(label); + self.cmd_buffer.commands.push(C::InsertDebugMarker(range)); + } + unsafe fn begin_debug_marker(&mut self, group_label: &str) { + let range = self.cmd_buffer.add_marker(group_label); + self.cmd_buffer.commands.push(C::PushDebugGroup(range)); + } + unsafe fn end_debug_marker(&mut self) { + self.cmd_buffer.commands.push(C::PopDebugGroup); + } + + unsafe fn set_render_pipeline(&mut self, pipeline: &super::RenderPipeline) { + self.state.topology = conv::map_primitive_topology(pipeline.primitive.topology); + + if self + .private_caps + .contains(super::PrivateCapabilities::VERTEX_BUFFER_LAYOUT) + { + for vat in pipeline.vertex_attributes.iter() { + let vb = &pipeline.vertex_buffers[vat.buffer_index as usize]; + // set the layout + self.cmd_buffer.commands.push(C::SetVertexAttribute { + buffer: None, + buffer_desc: vb.clone(), + attribute_desc: vat.clone(), + }); + } + } else { + for vat in &self.state.vertex_attributes { + self.cmd_buffer + .commands + .push(C::UnsetVertexAttribute(vat.location)); + } + self.state.vertex_attributes.clear(); + + self.state.dirty_vbuf_mask = 0; + // copy vertex attributes + for vat in pipeline.vertex_attributes.iter() { + //Note: we can invalidate more carefully here. + self.state.dirty_vbuf_mask |= 1 << vat.buffer_index; + self.state.vertex_attributes.push(vat.clone()); + } + } + + self.state.instance_vbuf_mask = 0; + // copy vertex state + for (index, (&mut (ref mut state_desc, _), pipe_desc)) in self + .state + .vertex_buffers + .iter_mut() + .zip(pipeline.vertex_buffers.iter()) + .enumerate() + { + if pipe_desc.step == wgt::VertexStepMode::Instance { + self.state.instance_vbuf_mask |= 1 << index; + } + if state_desc != pipe_desc { + self.state.dirty_vbuf_mask |= 1 << index; + *state_desc = pipe_desc.clone(); + } + } + + self.set_pipeline_inner(&pipeline.inner); + + // set primitive state + let prim_state = conv::map_primitive_state(&pipeline.primitive); + if prim_state != self.state.primitive { + self.cmd_buffer + .commands + .push(C::SetPrimitive(prim_state.clone())); + self.state.primitive = prim_state; + } + + // set depth/stencil states + let mut aspects = crate::FormatAspects::empty(); + if pipeline.depth_bias != self.state.depth_bias { + self.state.depth_bias = pipeline.depth_bias; + self.cmd_buffer + .commands + .push(C::SetDepthBias(pipeline.depth_bias)); + } + if let Some(ref depth) = pipeline.depth { + aspects |= crate::FormatAspects::DEPTH; + self.cmd_buffer.commands.push(C::SetDepth(depth.clone())); + } + if let Some(ref stencil) = pipeline.stencil { + aspects |= crate::FormatAspects::STENCIL; + self.state.stencil = stencil.clone(); + self.rebind_stencil_func(); + if stencil.front.ops == stencil.back.ops + && stencil.front.mask_write == stencil.back.mask_write + { + self.cmd_buffer.commands.push(C::SetStencilOps { + face: glow::FRONT_AND_BACK, + write_mask: stencil.front.mask_write, + ops: stencil.front.ops.clone(), + }); + } else { + self.cmd_buffer.commands.push(C::SetStencilOps { + face: glow::FRONT, + write_mask: stencil.front.mask_write, + ops: stencil.front.ops.clone(), + }); + self.cmd_buffer.commands.push(C::SetStencilOps { + face: glow::BACK, + write_mask: stencil.back.mask_write, + ops: stencil.back.ops.clone(), + }); + } + } + self.cmd_buffer + .commands + .push(C::ConfigureDepthStencil(aspects)); + + // set multisampling state + if pipeline.alpha_to_coverage_enabled != self.state.alpha_to_coverage_enabled { + self.state.alpha_to_coverage_enabled = pipeline.alpha_to_coverage_enabled; + self.cmd_buffer + .commands + .push(C::SetAlphaToCoverage(pipeline.alpha_to_coverage_enabled)); + } + + // set blend states + if self.state.color_targets[..] != pipeline.color_targets[..] { + if pipeline + .color_targets + .iter() + .skip(1) + .any(|ct| *ct != pipeline.color_targets[0]) + { + for (index, ct) in pipeline.color_targets.iter().enumerate() { + self.cmd_buffer.commands.push(C::SetColorTarget { + draw_buffer_index: Some(index as u32), + desc: ct.clone(), + }); + } + } else { + self.cmd_buffer.commands.push(C::SetColorTarget { + draw_buffer_index: None, + desc: pipeline.color_targets.first().cloned().unwrap_or_default(), + }); + } + } + self.state.color_targets.clear(); + for ct in pipeline.color_targets.iter() { + self.state.color_targets.push(ct.clone()); + } + + // set clip plane count + if pipeline.inner.clip_distance_count != self.state.clip_distance_count { + self.cmd_buffer.commands.push(C::SetClipDistances { + old_count: self.state.clip_distance_count, + new_count: pipeline.inner.clip_distance_count, + }); + self.state.clip_distance_count = pipeline.inner.clip_distance_count; + } + } + + unsafe fn set_index_buffer<'a>( + &mut self, + binding: crate::BufferBinding<'a, super::Buffer>, + format: wgt::IndexFormat, + ) { + self.state.index_offset = binding.offset; + self.state.index_format = format; + self.cmd_buffer + .commands + .push(C::SetIndexBuffer(binding.buffer.raw.unwrap())); + } + unsafe fn set_vertex_buffer<'a>( + &mut self, + index: u32, + binding: crate::BufferBinding<'a, super::Buffer>, + ) { + self.state.dirty_vbuf_mask |= 1 << index; + let (_, ref mut vb) = self.state.vertex_buffers[index as usize]; + *vb = Some(super::BufferBinding { + raw: binding.buffer.raw.unwrap(), + offset: binding.offset, + }); + } + unsafe fn set_viewport(&mut self, rect: &crate::Rect, depth: Range) { + self.cmd_buffer.commands.push(C::SetViewport { + rect: crate::Rect { + x: rect.x as i32, + y: rect.y as i32, + w: rect.w as i32, + h: rect.h as i32, + }, + depth, + }); + } + unsafe fn set_scissor_rect(&mut self, rect: &crate::Rect) { + self.cmd_buffer.commands.push(C::SetScissor(crate::Rect { + x: rect.x as i32, + y: rect.y as i32, + w: rect.w as i32, + h: rect.h as i32, + })); + } + unsafe fn set_stencil_reference(&mut self, value: u32) { + self.state.stencil.front.reference = value; + self.state.stencil.back.reference = value; + self.rebind_stencil_func(); + } + unsafe fn set_blend_constants(&mut self, color: &[f32; 4]) { + self.cmd_buffer.commands.push(C::SetBlendConstant(*color)); + } + + unsafe fn draw( + &mut self, + first_vertex: u32, + vertex_count: u32, + first_instance: u32, + instance_count: u32, + ) { + self.prepare_draw(first_instance); + #[allow(clippy::clone_on_copy)] // False positive when cloning glow::UniformLocation + self.cmd_buffer.commands.push(C::Draw { + topology: self.state.topology, + first_vertex, + vertex_count, + first_instance, + instance_count, + first_instance_location: self.state.first_instance_location.clone(), + }); + } + unsafe fn draw_indexed( + &mut self, + first_index: u32, + index_count: u32, + base_vertex: i32, + first_instance: u32, + instance_count: u32, + ) { + self.prepare_draw(first_instance); + let (index_size, index_type) = match self.state.index_format { + wgt::IndexFormat::Uint16 => (2, glow::UNSIGNED_SHORT), + wgt::IndexFormat::Uint32 => (4, glow::UNSIGNED_INT), + }; + let index_offset = self.state.index_offset + index_size * first_index as wgt::BufferAddress; + #[allow(clippy::clone_on_copy)] // False positive when cloning glow::UniformLocation + self.cmd_buffer.commands.push(C::DrawIndexed { + topology: self.state.topology, + index_type, + index_offset, + index_count, + base_vertex, + first_instance, + instance_count, + first_instance_location: self.state.first_instance_location.clone(), + }); + } + unsafe fn draw_mesh_tasks( + &mut self, + _group_count_x: u32, + _group_count_y: u32, + _group_count_z: u32, + ) { + unreachable!() + } + unsafe fn draw_indirect( + &mut self, + buffer: &super::Buffer, + offset: wgt::BufferAddress, + draw_count: u32, + ) { + self.prepare_draw(0); + for draw in 0..draw_count as wgt::BufferAddress { + let indirect_offset = + offset + draw * size_of::() as wgt::BufferAddress; + #[allow(clippy::clone_on_copy)] // False positive when cloning glow::UniformLocation + self.cmd_buffer.commands.push(C::DrawIndirect { + topology: self.state.topology, + indirect_buf: buffer.raw.unwrap(), + indirect_offset, + first_instance_location: self.state.first_instance_location.clone(), + }); + } + } + unsafe fn draw_indexed_indirect( + &mut self, + buffer: &super::Buffer, + offset: wgt::BufferAddress, + draw_count: u32, + ) { + self.prepare_draw(0); + let index_type = match self.state.index_format { + wgt::IndexFormat::Uint16 => glow::UNSIGNED_SHORT, + wgt::IndexFormat::Uint32 => glow::UNSIGNED_INT, + }; + for draw in 0..draw_count as wgt::BufferAddress { + let indirect_offset = + offset + draw * size_of::() as wgt::BufferAddress; + #[allow(clippy::clone_on_copy)] // False positive when cloning glow::UniformLocation + self.cmd_buffer.commands.push(C::DrawIndexedIndirect { + topology: self.state.topology, + index_type, + indirect_buf: buffer.raw.unwrap(), + indirect_offset, + first_instance_location: self.state.first_instance_location.clone(), + }); + } + } + unsafe fn draw_mesh_tasks_indirect( + &mut self, + _buffer: &::Buffer, + _offset: wgt::BufferAddress, + _draw_count: u32, + ) { + unreachable!() + } + unsafe fn draw_indirect_count( + &mut self, + _buffer: &super::Buffer, + _offset: wgt::BufferAddress, + _count_buffer: &super::Buffer, + _count_offset: wgt::BufferAddress, + _max_count: u32, + ) { + unreachable!() + } + unsafe fn draw_indexed_indirect_count( + &mut self, + _buffer: &super::Buffer, + _offset: wgt::BufferAddress, + _count_buffer: &super::Buffer, + _count_offset: wgt::BufferAddress, + _max_count: u32, + ) { + unreachable!() + } + unsafe fn draw_mesh_tasks_indirect_count( + &mut self, + _buffer: &::Buffer, + _offset: wgt::BufferAddress, + _count_buffer: &::Buffer, + _count_offset: wgt::BufferAddress, + _max_count: u32, + ) { + unreachable!() + } + + // compute + + unsafe fn begin_compute_pass(&mut self, desc: &crate::ComputePassDescriptor) { + debug_assert!(self.state.end_of_pass_timestamp.is_none()); + if let Some(ref t) = desc.timestamp_writes { + if let Some(index) = t.beginning_of_pass_write_index { + unsafe { self.write_timestamp(t.query_set, index) } + } + self.state.end_of_pass_timestamp = t + .end_of_pass_write_index + .map(|index| t.query_set.queries[index as usize]); + } + + if let Some(label) = desc.label { + let range = self.cmd_buffer.add_marker(label); + self.cmd_buffer.commands.push(C::PushDebugGroup(range)); + self.state.has_pass_label = true; + } + } + unsafe fn end_compute_pass(&mut self) { + if self.state.has_pass_label { + self.cmd_buffer.commands.push(C::PopDebugGroup); + self.state.has_pass_label = false; + } + + if let Some(query) = self.state.end_of_pass_timestamp.take() { + self.cmd_buffer.commands.push(C::TimestampQuery(query)); + } + } + + unsafe fn set_compute_pipeline(&mut self, pipeline: &super::ComputePipeline) { + self.set_pipeline_inner(&pipeline.inner); + } + + unsafe fn dispatch(&mut self, count: [u32; 3]) { + // Empty dispatches are invalid in OpenGL, but valid in WebGPU. + if count.contains(&0) { + return; + } + self.cmd_buffer.commands.push(C::Dispatch(count)); + } + unsafe fn dispatch_indirect(&mut self, buffer: &super::Buffer, offset: wgt::BufferAddress) { + self.cmd_buffer.commands.push(C::DispatchIndirect { + indirect_buf: buffer.raw.unwrap(), + indirect_offset: offset, + }); + } + + unsafe fn build_acceleration_structures<'a, T>( + &mut self, + _descriptor_count: u32, + _descriptors: T, + ) where + super::Api: 'a, + T: IntoIterator< + Item = crate::BuildAccelerationStructureDescriptor< + 'a, + super::Buffer, + super::AccelerationStructure, + >, + >, + { + unimplemented!() + } + + unsafe fn place_acceleration_structure_barrier( + &mut self, + _barriers: crate::AccelerationStructureBarrier, + ) { + unimplemented!() + } + + unsafe fn copy_acceleration_structure_to_acceleration_structure( + &mut self, + _src: &super::AccelerationStructure, + _dst: &super::AccelerationStructure, + _copy: wgt::AccelerationStructureCopy, + ) { + unimplemented!() + } + + unsafe fn read_acceleration_structure_compact_size( + &mut self, + _acceleration_structure: &super::AccelerationStructure, + _buf: &super::Buffer, + ) { + unimplemented!() + } + + unsafe fn set_acceleration_structure_dependencies( + _command_buffers: &[&super::CommandBuffer], + _dependencies: &[&super::AccelerationStructure], + ) { + unimplemented!() + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/gles/conv.rs b/third_party/wgpu-hal-29.0.4/src/gles/conv.rs new file mode 100644 index 0000000..5b54f8f --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/conv.rs @@ -0,0 +1,430 @@ +impl super::AdapterShared { + pub(super) fn describe_texture_format( + &self, + texture_format: wgt::TextureFormat, + ) -> super::TextureFormatDesc { + use wgt::TextureFormat as Tf; + use wgt::{AstcBlock, AstcChannel}; + + let (internal, external, data_type) = match texture_format { + Tf::R8Unorm => (glow::R8, glow::RED, glow::UNSIGNED_BYTE), + Tf::R8Snorm => (glow::R8_SNORM, glow::RED, glow::BYTE), + Tf::R8Uint => (glow::R8UI, glow::RED_INTEGER, glow::UNSIGNED_BYTE), + Tf::R8Sint => (glow::R8I, glow::RED_INTEGER, glow::BYTE), + Tf::R16Uint => (glow::R16UI, glow::RED_INTEGER, glow::UNSIGNED_SHORT), + Tf::R16Sint => (glow::R16I, glow::RED_INTEGER, glow::SHORT), + Tf::R16Unorm => (glow::R16, glow::RED, glow::UNSIGNED_SHORT), + Tf::R16Snorm => (glow::R16_SNORM, glow::RED, glow::SHORT), + Tf::R16Float => (glow::R16F, glow::RED, glow::HALF_FLOAT), + Tf::Rg8Unorm => (glow::RG8, glow::RG, glow::UNSIGNED_BYTE), + Tf::Rg8Snorm => (glow::RG8_SNORM, glow::RG, glow::BYTE), + Tf::Rg8Uint => (glow::RG8UI, glow::RG_INTEGER, glow::UNSIGNED_BYTE), + Tf::Rg8Sint => (glow::RG8I, glow::RG_INTEGER, glow::BYTE), + Tf::R32Uint => (glow::R32UI, glow::RED_INTEGER, glow::UNSIGNED_INT), + Tf::R32Sint => (glow::R32I, glow::RED_INTEGER, glow::INT), + Tf::R32Float => (glow::R32F, glow::RED, glow::FLOAT), + Tf::Rg16Uint => (glow::RG16UI, glow::RG_INTEGER, glow::UNSIGNED_SHORT), + Tf::Rg16Sint => (glow::RG16I, glow::RG_INTEGER, glow::SHORT), + Tf::Rg16Unorm => (glow::RG16, glow::RG, glow::UNSIGNED_SHORT), + Tf::Rg16Snorm => (glow::RG16_SNORM, glow::RG, glow::SHORT), + Tf::Rg16Float => (glow::RG16F, glow::RG, glow::HALF_FLOAT), + Tf::Rgba8Unorm => (glow::RGBA8, glow::RGBA, glow::UNSIGNED_BYTE), + Tf::Rgba8UnormSrgb => (glow::SRGB8_ALPHA8, glow::RGBA, glow::UNSIGNED_BYTE), + Tf::Bgra8UnormSrgb => (glow::SRGB8_ALPHA8, glow::BGRA, glow::UNSIGNED_BYTE), //TODO? + Tf::Rgba8Snorm => (glow::RGBA8_SNORM, glow::RGBA, glow::BYTE), + Tf::Bgra8Unorm => (glow::RGBA8, glow::BGRA, glow::UNSIGNED_BYTE), //TODO? + Tf::Rgba8Uint => (glow::RGBA8UI, glow::RGBA_INTEGER, glow::UNSIGNED_BYTE), + Tf::Rgba8Sint => (glow::RGBA8I, glow::RGBA_INTEGER, glow::BYTE), + Tf::Rgb10a2Uint => ( + glow::RGB10_A2UI, + glow::RGBA_INTEGER, + glow::UNSIGNED_INT_2_10_10_10_REV, + ), + Tf::Rgb10a2Unorm => ( + glow::RGB10_A2, + glow::RGBA, + glow::UNSIGNED_INT_2_10_10_10_REV, + ), + Tf::Rg11b10Ufloat => ( + glow::R11F_G11F_B10F, + glow::RGB, + glow::UNSIGNED_INT_10F_11F_11F_REV, + ), + Tf::R64Uint => (glow::RG32UI, glow::RED_INTEGER, glow::UNSIGNED_INT), + Tf::Rg32Uint => (glow::RG32UI, glow::RG_INTEGER, glow::UNSIGNED_INT), + Tf::Rg32Sint => (glow::RG32I, glow::RG_INTEGER, glow::INT), + Tf::Rg32Float => (glow::RG32F, glow::RG, glow::FLOAT), + Tf::Rgba16Uint => (glow::RGBA16UI, glow::RGBA_INTEGER, glow::UNSIGNED_SHORT), + Tf::Rgba16Sint => (glow::RGBA16I, glow::RGBA_INTEGER, glow::SHORT), + Tf::Rgba16Unorm => (glow::RGBA16, glow::RGBA, glow::UNSIGNED_SHORT), + Tf::Rgba16Snorm => (glow::RGBA16_SNORM, glow::RGBA, glow::SHORT), + Tf::Rgba16Float => (glow::RGBA16F, glow::RGBA, glow::HALF_FLOAT), + Tf::Rgba32Uint => (glow::RGBA32UI, glow::RGBA_INTEGER, glow::UNSIGNED_INT), + Tf::Rgba32Sint => (glow::RGBA32I, glow::RGBA_INTEGER, glow::INT), + Tf::Rgba32Float => (glow::RGBA32F, glow::RGBA, glow::FLOAT), + Tf::Stencil8 => ( + glow::STENCIL_INDEX8, + glow::STENCIL_INDEX, + glow::UNSIGNED_BYTE, + ), + Tf::Depth16Unorm => ( + glow::DEPTH_COMPONENT16, + glow::DEPTH_COMPONENT, + glow::UNSIGNED_SHORT, + ), + Tf::Depth32Float => (glow::DEPTH_COMPONENT32F, glow::DEPTH_COMPONENT, glow::FLOAT), + Tf::Depth32FloatStencil8 => ( + glow::DEPTH32F_STENCIL8, + glow::DEPTH_STENCIL, + glow::FLOAT_32_UNSIGNED_INT_24_8_REV, + ), + Tf::Depth24Plus => ( + glow::DEPTH_COMPONENT24, + glow::DEPTH_COMPONENT, + glow::UNSIGNED_INT, + ), + Tf::Depth24PlusStencil8 => ( + glow::DEPTH24_STENCIL8, + glow::DEPTH_STENCIL, + glow::UNSIGNED_INT_24_8, + ), + Tf::NV12 => unreachable!(), + Tf::P010 => unreachable!(), + Tf::Rgb9e5Ufloat => (glow::RGB9_E5, glow::RGB, glow::UNSIGNED_INT_5_9_9_9_REV), + Tf::Bc1RgbaUnorm => (glow::COMPRESSED_RGBA_S3TC_DXT1_EXT, glow::RGBA, 0), + Tf::Bc1RgbaUnormSrgb => (glow::COMPRESSED_SRGB_ALPHA_S3TC_DXT1_EXT, glow::RGBA, 0), + Tf::Bc2RgbaUnorm => (glow::COMPRESSED_RGBA_S3TC_DXT3_EXT, glow::RGBA, 0), + Tf::Bc2RgbaUnormSrgb => (glow::COMPRESSED_SRGB_ALPHA_S3TC_DXT3_EXT, glow::RGBA, 0), + Tf::Bc3RgbaUnorm => (glow::COMPRESSED_RGBA_S3TC_DXT5_EXT, glow::RGBA, 0), + Tf::Bc3RgbaUnormSrgb => (glow::COMPRESSED_SRGB_ALPHA_S3TC_DXT5_EXT, glow::RGBA, 0), + Tf::Bc4RUnorm => (glow::COMPRESSED_RED_RGTC1, glow::RED, 0), + Tf::Bc4RSnorm => (glow::COMPRESSED_SIGNED_RED_RGTC1, glow::RED, 0), + Tf::Bc5RgUnorm => (glow::COMPRESSED_RG_RGTC2, glow::RG, 0), + Tf::Bc5RgSnorm => (glow::COMPRESSED_SIGNED_RG_RGTC2, glow::RG, 0), + Tf::Bc6hRgbUfloat => (glow::COMPRESSED_RGB_BPTC_UNSIGNED_FLOAT, glow::RGB, 0), + Tf::Bc6hRgbFloat => (glow::COMPRESSED_RGB_BPTC_SIGNED_FLOAT, glow::RGB, 0), + Tf::Bc7RgbaUnorm => (glow::COMPRESSED_RGBA_BPTC_UNORM, glow::RGBA, 0), + Tf::Bc7RgbaUnormSrgb => (glow::COMPRESSED_SRGB_ALPHA_BPTC_UNORM, glow::RGBA, 0), + Tf::Etc2Rgb8Unorm => (glow::COMPRESSED_RGB8_ETC2, glow::RGB, 0), + Tf::Etc2Rgb8UnormSrgb => (glow::COMPRESSED_SRGB8_ETC2, glow::RGB, 0), + Tf::Etc2Rgb8A1Unorm => ( + glow::COMPRESSED_RGB8_PUNCHTHROUGH_ALPHA1_ETC2, + glow::RGBA, + 0, + ), + Tf::Etc2Rgb8A1UnormSrgb => ( + glow::COMPRESSED_SRGB8_PUNCHTHROUGH_ALPHA1_ETC2, + glow::RGBA, + 0, + ), + Tf::Etc2Rgba8Unorm => (glow::COMPRESSED_RGBA8_ETC2_EAC, glow::RGBA, 0), + Tf::Etc2Rgba8UnormSrgb => (glow::COMPRESSED_SRGB8_ALPHA8_ETC2_EAC, glow::RGBA, 0), + Tf::EacR11Unorm => (glow::COMPRESSED_R11_EAC, glow::RED, 0), + Tf::EacR11Snorm => (glow::COMPRESSED_SIGNED_R11_EAC, glow::RED, 0), + Tf::EacRg11Unorm => (glow::COMPRESSED_RG11_EAC, glow::RG, 0), + Tf::EacRg11Snorm => (glow::COMPRESSED_SIGNED_RG11_EAC, glow::RG, 0), + Tf::Astc { block, channel } => match channel { + AstcChannel::Unorm | AstcChannel::Hdr => match block { + AstcBlock::B4x4 => (glow::COMPRESSED_RGBA_ASTC_4x4_KHR, glow::RGBA, 0), + AstcBlock::B5x4 => (glow::COMPRESSED_RGBA_ASTC_5x4_KHR, glow::RGBA, 0), + AstcBlock::B5x5 => (glow::COMPRESSED_RGBA_ASTC_5x5_KHR, glow::RGBA, 0), + AstcBlock::B6x5 => (glow::COMPRESSED_RGBA_ASTC_6x5_KHR, glow::RGBA, 0), + AstcBlock::B6x6 => (glow::COMPRESSED_RGBA_ASTC_6x6_KHR, glow::RGBA, 0), + AstcBlock::B8x5 => (glow::COMPRESSED_RGBA_ASTC_8x5_KHR, glow::RGBA, 0), + AstcBlock::B8x6 => (glow::COMPRESSED_RGBA_ASTC_8x6_KHR, glow::RGBA, 0), + AstcBlock::B8x8 => (glow::COMPRESSED_RGBA_ASTC_8x8_KHR, glow::RGBA, 0), + AstcBlock::B10x5 => (glow::COMPRESSED_RGBA_ASTC_10x5_KHR, glow::RGBA, 0), + AstcBlock::B10x6 => (glow::COMPRESSED_RGBA_ASTC_10x6_KHR, glow::RGBA, 0), + AstcBlock::B10x8 => (glow::COMPRESSED_RGBA_ASTC_10x8_KHR, glow::RGBA, 0), + AstcBlock::B10x10 => (glow::COMPRESSED_RGBA_ASTC_10x10_KHR, glow::RGBA, 0), + AstcBlock::B12x10 => (glow::COMPRESSED_RGBA_ASTC_12x10_KHR, glow::RGBA, 0), + AstcBlock::B12x12 => (glow::COMPRESSED_RGBA_ASTC_12x12_KHR, glow::RGBA, 0), + }, + AstcChannel::UnormSrgb => match block { + AstcBlock::B4x4 => (glow::COMPRESSED_SRGB8_ALPHA8_ASTC_4x4_KHR, glow::RGBA, 0), + AstcBlock::B5x4 => (glow::COMPRESSED_SRGB8_ALPHA8_ASTC_5x4_KHR, glow::RGBA, 0), + AstcBlock::B5x5 => (glow::COMPRESSED_SRGB8_ALPHA8_ASTC_5x5_KHR, glow::RGBA, 0), + AstcBlock::B6x5 => (glow::COMPRESSED_SRGB8_ALPHA8_ASTC_6x5_KHR, glow::RGBA, 0), + AstcBlock::B6x6 => (glow::COMPRESSED_SRGB8_ALPHA8_ASTC_6x6_KHR, glow::RGBA, 0), + AstcBlock::B8x5 => (glow::COMPRESSED_SRGB8_ALPHA8_ASTC_8x5_KHR, glow::RGBA, 0), + AstcBlock::B8x6 => (glow::COMPRESSED_SRGB8_ALPHA8_ASTC_8x6_KHR, glow::RGBA, 0), + AstcBlock::B8x8 => (glow::COMPRESSED_SRGB8_ALPHA8_ASTC_8x8_KHR, glow::RGBA, 0), + AstcBlock::B10x5 => { + (glow::COMPRESSED_SRGB8_ALPHA8_ASTC_10x5_KHR, glow::RGBA, 0) + } + AstcBlock::B10x6 => { + (glow::COMPRESSED_SRGB8_ALPHA8_ASTC_10x6_KHR, glow::RGBA, 0) + } + AstcBlock::B10x8 => { + (glow::COMPRESSED_SRGB8_ALPHA8_ASTC_10x8_KHR, glow::RGBA, 0) + } + AstcBlock::B10x10 => { + (glow::COMPRESSED_SRGB8_ALPHA8_ASTC_10x10_KHR, glow::RGBA, 0) + } + AstcBlock::B12x10 => { + (glow::COMPRESSED_SRGB8_ALPHA8_ASTC_12x10_KHR, glow::RGBA, 0) + } + AstcBlock::B12x12 => { + (glow::COMPRESSED_SRGB8_ALPHA8_ASTC_12x12_KHR, glow::RGBA, 0) + } + }, + }, + }; + + super::TextureFormatDesc { + internal, + external, + data_type, + } + } +} + +pub(super) fn describe_vertex_format(vertex_format: wgt::VertexFormat) -> super::VertexFormatDesc { + use super::VertexAttribKind as Vak; + use wgt::VertexFormat as Vf; + + let (element_count, element_format, attrib_kind) = match vertex_format { + Vf::Unorm8 => (1, glow::UNSIGNED_BYTE, Vak::Float), + Vf::Snorm8 => (1, glow::BYTE, Vak::Float), + Vf::Uint8 => (1, glow::UNSIGNED_BYTE, Vak::Integer), + Vf::Sint8 => (1, glow::BYTE, Vak::Integer), + Vf::Unorm8x2 => (2, glow::UNSIGNED_BYTE, Vak::Float), + Vf::Snorm8x2 => (2, glow::BYTE, Vak::Float), + Vf::Uint8x2 => (2, glow::UNSIGNED_BYTE, Vak::Integer), + Vf::Sint8x2 => (2, glow::BYTE, Vak::Integer), + Vf::Unorm8x4 => (4, glow::UNSIGNED_BYTE, Vak::Float), + Vf::Snorm8x4 => (4, glow::BYTE, Vak::Float), + Vf::Uint8x4 => (4, glow::UNSIGNED_BYTE, Vak::Integer), + Vf::Sint8x4 => (4, glow::BYTE, Vak::Integer), + Vf::Unorm16 => (1, glow::UNSIGNED_SHORT, Vak::Float), + Vf::Snorm16 => (1, glow::SHORT, Vak::Float), + Vf::Uint16 => (1, glow::UNSIGNED_SHORT, Vak::Integer), + Vf::Sint16 => (1, glow::SHORT, Vak::Integer), + Vf::Float16 => (1, glow::HALF_FLOAT, Vak::Float), + Vf::Unorm16x2 => (2, glow::UNSIGNED_SHORT, Vak::Float), + Vf::Snorm16x2 => (2, glow::SHORT, Vak::Float), + Vf::Uint16x2 => (2, glow::UNSIGNED_SHORT, Vak::Integer), + Vf::Sint16x2 => (2, glow::SHORT, Vak::Integer), + Vf::Float16x2 => (2, glow::HALF_FLOAT, Vak::Float), + Vf::Unorm16x4 => (4, glow::UNSIGNED_SHORT, Vak::Float), + Vf::Snorm16x4 => (4, glow::SHORT, Vak::Float), + Vf::Uint16x4 => (4, glow::UNSIGNED_SHORT, Vak::Integer), + Vf::Sint16x4 => (4, glow::SHORT, Vak::Integer), + Vf::Float16x4 => (4, glow::HALF_FLOAT, Vak::Float), + Vf::Uint32 => (1, glow::UNSIGNED_INT, Vak::Integer), + Vf::Sint32 => (1, glow::INT, Vak::Integer), + Vf::Float32 => (1, glow::FLOAT, Vak::Float), + Vf::Uint32x2 => (2, glow::UNSIGNED_INT, Vak::Integer), + Vf::Sint32x2 => (2, glow::INT, Vak::Integer), + Vf::Float32x2 => (2, glow::FLOAT, Vak::Float), + Vf::Uint32x3 => (3, glow::UNSIGNED_INT, Vak::Integer), + Vf::Sint32x3 => (3, glow::INT, Vak::Integer), + Vf::Float32x3 => (3, glow::FLOAT, Vak::Float), + Vf::Uint32x4 => (4, glow::UNSIGNED_INT, Vak::Integer), + Vf::Sint32x4 => (4, glow::INT, Vak::Integer), + Vf::Float32x4 => (4, glow::FLOAT, Vak::Float), + Vf::Unorm10_10_10_2 => (4, glow::UNSIGNED_INT_2_10_10_10_REV, Vak::Float), + Vf::Unorm8x4Bgra => (glow::BGRA as i32, glow::UNSIGNED_BYTE, Vak::Float), + Vf::Float64 | Vf::Float64x2 | Vf::Float64x3 | Vf::Float64x4 => unimplemented!(), + }; + + super::VertexFormatDesc { + element_count, + element_format, + attrib_kind, + } +} + +pub fn map_filter_modes( + min: wgt::FilterMode, + mag: wgt::FilterMode, + mip: wgt::MipmapFilterMode, +) -> (u32, u32) { + use wgt::FilterMode as Fm; + use wgt::MipmapFilterMode as Mfm; + + let mag_filter = match mag { + Fm::Nearest => glow::NEAREST, + Fm::Linear => glow::LINEAR, + }; + + let min_filter = match (min, mip) { + (Fm::Nearest, Mfm::Nearest) => glow::NEAREST_MIPMAP_NEAREST, + (Fm::Nearest, Mfm::Linear) => glow::NEAREST_MIPMAP_LINEAR, + (Fm::Linear, Mfm::Nearest) => glow::LINEAR_MIPMAP_NEAREST, + (Fm::Linear, Mfm::Linear) => glow::LINEAR_MIPMAP_LINEAR, + }; + + (min_filter, mag_filter) +} + +pub fn map_address_mode(mode: wgt::AddressMode) -> u32 { + match mode { + wgt::AddressMode::Repeat => glow::REPEAT, + wgt::AddressMode::MirrorRepeat => glow::MIRRORED_REPEAT, + wgt::AddressMode::ClampToEdge => glow::CLAMP_TO_EDGE, + wgt::AddressMode::ClampToBorder => glow::CLAMP_TO_BORDER, + //wgt::AddressMode::MirrorClamp => glow::MIRROR_CLAMP_TO_EDGE, + } +} + +pub fn map_compare_func(fun: wgt::CompareFunction) -> u32 { + use wgt::CompareFunction as Cf; + match fun { + Cf::Never => glow::NEVER, + Cf::Less => glow::LESS, + Cf::LessEqual => glow::LEQUAL, + Cf::Equal => glow::EQUAL, + Cf::GreaterEqual => glow::GEQUAL, + Cf::Greater => glow::GREATER, + Cf::NotEqual => glow::NOTEQUAL, + Cf::Always => glow::ALWAYS, + } +} + +pub fn map_primitive_topology(topology: wgt::PrimitiveTopology) -> u32 { + use wgt::PrimitiveTopology as Pt; + match topology { + Pt::PointList => glow::POINTS, + Pt::LineList => glow::LINES, + Pt::LineStrip => glow::LINE_STRIP, + Pt::TriangleList => glow::TRIANGLES, + Pt::TriangleStrip => glow::TRIANGLE_STRIP, + } +} + +pub(super) fn map_primitive_state(state: &wgt::PrimitiveState) -> super::PrimitiveState { + super::PrimitiveState { + //Note: we are flipping the front face, so that + // the Y-flip in the generated GLSL keeps the same visibility. + // See `naga::back::glsl::WriterFlags::ADJUST_COORDINATE_SPACE`. + front_face: match state.front_face { + wgt::FrontFace::Cw => glow::CCW, + wgt::FrontFace::Ccw => glow::CW, + }, + cull_face: match state.cull_mode { + Some(wgt::Face::Front) => glow::FRONT, + Some(wgt::Face::Back) => glow::BACK, + None => 0, + }, + unclipped_depth: state.unclipped_depth, + polygon_mode: match state.polygon_mode { + wgt::PolygonMode::Fill => glow::FILL, + wgt::PolygonMode::Line => glow::LINE, + wgt::PolygonMode::Point => glow::POINT, + }, + } +} + +pub fn _map_view_dimension(dim: wgt::TextureViewDimension) -> u32 { + use wgt::TextureViewDimension as Tvd; + match dim { + Tvd::D1 | Tvd::D2 => glow::TEXTURE_2D, + Tvd::D2Array => glow::TEXTURE_2D_ARRAY, + Tvd::Cube => glow::TEXTURE_CUBE_MAP, + Tvd::CubeArray => glow::TEXTURE_CUBE_MAP_ARRAY, + Tvd::D3 => glow::TEXTURE_3D, + } +} + +fn map_stencil_op(operation: wgt::StencilOperation) -> u32 { + use wgt::StencilOperation as So; + match operation { + So::Keep => glow::KEEP, + So::Zero => glow::ZERO, + So::Replace => glow::REPLACE, + So::Invert => glow::INVERT, + So::IncrementClamp => glow::INCR, + So::DecrementClamp => glow::DECR, + So::IncrementWrap => glow::INCR_WRAP, + So::DecrementWrap => glow::DECR_WRAP, + } +} + +fn map_stencil_ops(face: &wgt::StencilFaceState) -> super::StencilOps { + super::StencilOps { + pass: map_stencil_op(face.pass_op), + fail: map_stencil_op(face.fail_op), + depth_fail: map_stencil_op(face.depth_fail_op), + } +} + +pub(super) fn map_stencil(state: &wgt::StencilState) -> super::StencilState { + super::StencilState { + front: super::StencilSide { + function: map_compare_func(state.front.compare), + mask_read: state.read_mask, + mask_write: state.write_mask, + reference: 0, + ops: map_stencil_ops(&state.front), + }, + back: super::StencilSide { + function: map_compare_func(state.back.compare), + mask_read: state.read_mask, + mask_write: state.write_mask, + reference: 0, + ops: map_stencil_ops(&state.back), + }, + } +} + +fn map_blend_factor(factor: wgt::BlendFactor) -> u32 { + use wgt::BlendFactor as Bf; + match factor { + Bf::Zero => glow::ZERO, + Bf::One => glow::ONE, + Bf::Src => glow::SRC_COLOR, + Bf::OneMinusSrc => glow::ONE_MINUS_SRC_COLOR, + Bf::Dst => glow::DST_COLOR, + Bf::OneMinusDst => glow::ONE_MINUS_DST_COLOR, + Bf::SrcAlpha => glow::SRC_ALPHA, + Bf::OneMinusSrcAlpha => glow::ONE_MINUS_SRC_ALPHA, + Bf::DstAlpha => glow::DST_ALPHA, + Bf::OneMinusDstAlpha => glow::ONE_MINUS_DST_ALPHA, + Bf::Constant => glow::CONSTANT_COLOR, + Bf::OneMinusConstant => glow::ONE_MINUS_CONSTANT_COLOR, + Bf::SrcAlphaSaturated => glow::SRC_ALPHA_SATURATE, + Bf::Src1 => glow::SRC1_COLOR, + Bf::OneMinusSrc1 => glow::ONE_MINUS_SRC1_COLOR, + Bf::Src1Alpha => glow::SRC1_ALPHA, + Bf::OneMinusSrc1Alpha => glow::ONE_MINUS_SRC1_ALPHA, + } +} + +fn map_blend_component(component: &wgt::BlendComponent) -> super::BlendComponent { + super::BlendComponent { + src: map_blend_factor(component.src_factor), + dst: map_blend_factor(component.dst_factor), + equation: match component.operation { + wgt::BlendOperation::Add => glow::FUNC_ADD, + wgt::BlendOperation::Subtract => glow::FUNC_SUBTRACT, + wgt::BlendOperation::ReverseSubtract => glow::FUNC_REVERSE_SUBTRACT, + wgt::BlendOperation::Min => glow::MIN, + wgt::BlendOperation::Max => glow::MAX, + }, + } +} + +pub(super) fn map_blend(blend: &wgt::BlendState) -> super::BlendDesc { + super::BlendDesc { + color: map_blend_component(&blend.color), + alpha: map_blend_component(&blend.alpha), + } +} + +pub(super) fn map_storage_access(access: wgt::StorageTextureAccess) -> u32 { + match access { + wgt::StorageTextureAccess::ReadOnly => glow::READ_ONLY, + wgt::StorageTextureAccess::WriteOnly => glow::WRITE_ONLY, + wgt::StorageTextureAccess::ReadWrite => glow::READ_WRITE, + wgt::StorageTextureAccess::Atomic => glow::READ_WRITE, + } +} + +pub(super) fn is_layered_target(target: u32) -> bool { + match target { + glow::TEXTURE_2D | glow::TEXTURE_CUBE_MAP => false, + glow::TEXTURE_2D_ARRAY | glow::TEXTURE_CUBE_MAP_ARRAY | glow::TEXTURE_3D => true, + _ => unreachable!(), + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/gles/device.rs b/third_party/wgpu-hal-29.0.4/src/gles/device.rs new file mode 100644 index 0000000..c16a2ab --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/device.rs @@ -0,0 +1,1677 @@ +use alloc::{ + borrow::ToOwned, format, string::String, string::ToString as _, sync::Arc, vec, vec::Vec, +}; +use core::{cmp::max, convert::TryInto, num::NonZeroU32, ptr, sync::atomic::Ordering}; + +use arrayvec::ArrayVec; +use glow::HasContext; +use naga::FastHashMap; + +use super::{conv, lock, MaybeMutex, PrivateCapabilities}; +use crate::auxil::map_naga_stage; +use crate::TlasInstance; + +type ShaderStage<'a> = ( + naga::ShaderStage, + &'a crate::ProgrammableStage<'a, super::ShaderModule>, +); +type NameBindingMap = FastHashMap; + +struct CompilationContext<'a> { + layout: &'a super::PipelineLayout, + sampler_map: &'a mut super::SamplerBindMap, + name_binding_map: &'a mut NameBindingMap, + immediates_items: &'a mut Vec, + multiview_mask: Option, + clip_distance_count: &'a mut u32, +} + +impl CompilationContext<'_> { + fn consume_reflection( + self, + gl: &glow::Context, + module: &naga::Module, + ep_info: &naga::valid::FunctionInfo, + reflection_info: naga::back::glsl::ReflectionInfo, + naga_stage: naga::ShaderStage, + program: glow::Program, + ) { + for (handle, var) in module.global_variables.iter() { + if ep_info[handle].is_empty() { + continue; + } + let register = match var.space { + naga::AddressSpace::Uniform => super::BindingRegister::UniformBuffers, + naga::AddressSpace::Storage { .. } => super::BindingRegister::StorageBuffers, + _ => continue, + }; + + let br = var.binding.as_ref().unwrap(); + let slot = self.layout.get_slot(br); + + let name = match reflection_info.uniforms.get(&handle) { + Some(name) => name.clone(), + None => continue, + }; + log::trace!( + "Rebind buffer: {:?} -> {}, register={:?}, slot={}", + var.name.as_ref(), + &name, + register, + slot + ); + self.name_binding_map.insert(name, (register, slot)); + } + + for (name, mapping) in reflection_info.texture_mapping { + let var = &module.global_variables[mapping.texture]; + let register = match module.types[var.ty].inner { + naga::TypeInner::Image { + class: naga::ImageClass::Storage { .. }, + .. + } => super::BindingRegister::Images, + _ => super::BindingRegister::Textures, + }; + + let tex_br = var.binding.as_ref().unwrap(); + let texture_linear_index = self.layout.get_slot(tex_br); + + self.name_binding_map + .insert(name, (register, texture_linear_index)); + if let Some(sampler_handle) = mapping.sampler { + let sam_br = module.global_variables[sampler_handle] + .binding + .as_ref() + .unwrap(); + let sampler_linear_index = self.layout.get_slot(sam_br); + self.sampler_map[texture_linear_index as usize] = Some(sampler_linear_index); + } + } + + for (name, location) in reflection_info.varying { + match naga_stage { + naga::ShaderStage::Vertex => { + assert_eq!(location.index, 0); + unsafe { gl.bind_attrib_location(program, location.location, &name) } + } + naga::ShaderStage::Fragment => { + assert_eq!(location.index, 0); + unsafe { gl.bind_frag_data_location(program, location.location, &name) } + } + naga::ShaderStage::Compute => {} + naga::ShaderStage::Task + | naga::ShaderStage::Mesh + | naga::ShaderStage::RayGeneration + | naga::ShaderStage::AnyHit + | naga::ShaderStage::ClosestHit + | naga::ShaderStage::Miss => unreachable!(), + } + } + + *self.immediates_items = reflection_info.immediates_items; + + if naga_stage == naga::ShaderStage::Vertex { + *self.clip_distance_count = reflection_info.clip_distance_count; + } + } +} + +impl super::Device { + /// # Safety + /// + /// - `name` must be created respecting `desc` + /// - `name` must be a texture + /// - If `drop_callback` is [`None`], wgpu-hal will take ownership of the texture. If + /// `drop_callback` is [`Some`], the texture must be valid until the callback is called. + #[cfg(any(native, Emscripten))] + pub unsafe fn texture_from_raw( + &self, + name: NonZeroU32, + desc: &crate::TextureDescriptor, + drop_callback: Option, + ) -> super::Texture { + super::Texture { + inner: super::TextureInner::Texture { + raw: glow::NativeTexture(name), + target: super::Texture::get_info_from_desc(desc), + }, + drop_guard: crate::DropGuard::from_option(drop_callback), + mip_level_count: desc.mip_level_count, + array_layer_count: desc.array_layer_count(), + format: desc.format, + format_desc: self.shared.describe_texture_format(desc.format), + copy_size: desc.copy_extent(), + } + } + + /// # Safety + /// + /// - `name` must be created respecting `desc` + /// - `name` must be a renderbuffer + /// - If `drop_callback` is [`None`], wgpu-hal will take ownership of the renderbuffer. If + /// `drop_callback` is [`Some`], the renderbuffer must be valid until the callback is called. + #[cfg(any(native, Emscripten))] + pub unsafe fn texture_from_raw_renderbuffer( + &self, + name: NonZeroU32, + desc: &crate::TextureDescriptor, + drop_callback: Option, + ) -> super::Texture { + super::Texture { + inner: super::TextureInner::Renderbuffer { + raw: glow::NativeRenderbuffer(name), + }, + drop_guard: crate::DropGuard::from_option(drop_callback), + mip_level_count: desc.mip_level_count, + array_layer_count: desc.array_layer_count(), + format: desc.format, + format_desc: self.shared.describe_texture_format(desc.format), + copy_size: desc.copy_extent(), + } + } + + unsafe fn compile_shader( + gl: &glow::Context, + shader: &str, + naga_stage: naga::ShaderStage, + #[cfg_attr(target_arch = "wasm32", allow(unused))] label: Option<&str>, + ) -> Result { + let target = match naga_stage { + naga::ShaderStage::Vertex => glow::VERTEX_SHADER, + naga::ShaderStage::Fragment => glow::FRAGMENT_SHADER, + naga::ShaderStage::Compute => glow::COMPUTE_SHADER, + naga::ShaderStage::Task + | naga::ShaderStage::Mesh + | naga::ShaderStage::RayGeneration + | naga::ShaderStage::AnyHit + | naga::ShaderStage::ClosestHit + | naga::ShaderStage::Miss => unreachable!(), + }; + + let raw = unsafe { gl.create_shader(target) }.unwrap(); + #[cfg(native)] + if gl.supports_debug() { + let name = raw.0.get(); + unsafe { gl.object_label(glow::SHADER, name, label) }; + } + + unsafe { gl.shader_source(raw, shader) }; + unsafe { gl.compile_shader(raw) }; + + log::debug!("\tCompiled shader {raw:?}"); + + let compiled_ok = unsafe { gl.get_shader_compile_status(raw) }; + let msg = unsafe { gl.get_shader_info_log(raw) }; + if compiled_ok { + if !msg.is_empty() { + log::debug!("\tCompile message: {msg}"); + } + Ok(raw) + } else { + log::error!("\tShader compilation failed: {msg}"); + unsafe { gl.delete_shader(raw) }; + Err(crate::PipelineError::Linkage( + map_naga_stage(naga_stage), + msg, + )) + } + } + + fn create_shader( + gl: &glow::Context, + naga_stage: naga::ShaderStage, + stage: &crate::ProgrammableStage, + context: CompilationContext, + program: glow::Program, + ) -> Result { + use naga::back::glsl; + let pipeline_options = glsl::PipelineOptions { + shader_stage: naga_stage, + entry_point: stage.entry_point.to_owned(), + multiview: context + .multiview_mask + .map(|a| NonZeroU32::new(a.get().count_ones()).unwrap()), + }; + + let (module, info) = naga::back::pipeline_constants::process_overrides( + &stage.module.source.module, + &stage.module.source.info, + Some((naga_stage, stage.entry_point)), + stage.constants, + ) + .map_err(|e| { + let msg = format!("{e}"); + crate::PipelineError::PipelineConstants(map_naga_stage(naga_stage), msg) + })?; + + let entry_point_index = module + .entry_points + .iter() + .position(|ep| ep.name.as_str() == stage.entry_point) + .ok_or(crate::PipelineError::EntryPoint(naga_stage))?; + + use naga::proc::BoundsCheckPolicy; + // The image bounds checks require the TEXTURE_LEVELS feature available in GL core 4.3+. + let version = gl.version(); + let image_check = if !version.is_embedded && (version.major, version.minor) >= (4, 3) { + BoundsCheckPolicy::ReadZeroSkipWrite + } else { + BoundsCheckPolicy::Unchecked + }; + + // Other bounds check are either provided by glsl or not implemented yet. + let policies = naga::proc::BoundsCheckPolicies { + index: BoundsCheckPolicy::Unchecked, + buffer: BoundsCheckPolicy::Unchecked, + image_load: image_check, + binding_array: BoundsCheckPolicy::Unchecked, + }; + + let mut output = String::new(); + let needs_temp_options = stage.zero_initialize_workgroup_memory + != context.layout.naga_options.zero_initialize_workgroup_memory; + let mut temp_options; + let naga_options = if needs_temp_options { + // We use a conditional here, as cloning the naga_options could be expensive + // That is, we want to avoid doing that unless we cannot avoid it + temp_options = context.layout.naga_options.clone(); + temp_options.zero_initialize_workgroup_memory = stage.zero_initialize_workgroup_memory; + &temp_options + } else { + &context.layout.naga_options + }; + let mut writer = glsl::Writer::new( + &mut output, + &module, + &info, + naga_options, + &pipeline_options, + policies, + ) + .map_err(|e| { + let msg = format!("{e}"); + crate::PipelineError::Linkage(map_naga_stage(naga_stage), msg) + })?; + + let reflection_info = writer.write().map_err(|e| { + let msg = format!("{e}"); + crate::PipelineError::Linkage(map_naga_stage(naga_stage), msg) + })?; + + log::debug!("Naga generated shader:\n{output}"); + + context.consume_reflection( + gl, + &module, + info.get_entry_point(entry_point_index), + reflection_info, + naga_stage, + program, + ); + + unsafe { Self::compile_shader(gl, &output, naga_stage, stage.module.label.as_deref()) } + } + + unsafe fn create_pipeline<'a>( + &self, + gl: &glow::Context, + shaders: ArrayVec, { crate::MAX_CONCURRENT_SHADER_STAGES }>, + layout: &super::PipelineLayout, + #[cfg_attr(target_arch = "wasm32", allow(unused))] label: Option<&str>, + multiview_mask: Option, + ) -> Result, crate::PipelineError> { + let mut program_stages = ArrayVec::new(); + let group_to_binding_to_slot = layout + .group_infos + .iter() + .map(|group| group.as_ref().map(|group| group.binding_to_slot.clone())) + .collect::>(); + for &(naga_stage, stage) in &shaders { + program_stages.push(super::ProgramStage { + naga_stage: naga_stage.to_owned(), + shader_id: stage.module.id, + entry_point: stage.entry_point.to_owned(), + zero_initialize_workgroup_memory: stage.zero_initialize_workgroup_memory, + constant_hash: Self::create_constant_hash(stage), + }); + } + let mut guard = self + .shared + .program_cache + .try_lock() + .expect("Couldn't acquire program_cache lock"); + // This guard ensures that we can't accidentally destroy a program whilst we're about to reuse it + // The only place that destroys a pipeline is also locking on `program_cache` + let program = guard + .entry(super::ProgramCacheKey { + stages: program_stages, + group_to_binding_to_slot: group_to_binding_to_slot.into_boxed_slice(), + }) + .or_insert_with(|| unsafe { + Self::create_program( + gl, + shaders, + layout, + label, + multiview_mask, + self.shared.shading_language_version, + self.shared.private_caps, + ) + }) + .to_owned()?; + drop(guard); + + Ok(program) + } + + fn create_constant_hash(stage: &crate::ProgrammableStage) -> Vec { + let mut buf: Vec = Vec::new(); + + for (key, value) in stage.constants.iter() { + buf.extend_from_slice(key.as_bytes()); + buf.extend_from_slice(&value.to_ne_bytes()); + } + + buf + } + + unsafe fn create_program<'a>( + gl: &glow::Context, + shaders: ArrayVec, { crate::MAX_CONCURRENT_SHADER_STAGES }>, + layout: &super::PipelineLayout, + #[cfg_attr(target_arch = "wasm32", allow(unused))] label: Option<&str>, + multiview_mask: Option, + glsl_version: naga::back::glsl::Version, + private_caps: PrivateCapabilities, + ) -> Result, crate::PipelineError> { + let glsl_version = match glsl_version { + naga::back::glsl::Version::Embedded { version, .. } => format!("{version} es"), + naga::back::glsl::Version::Desktop(version) => format!("{version}"), + }; + let program = unsafe { gl.create_program() }.unwrap(); + #[cfg(native)] + if let Some(label) = label { + if private_caps.contains(PrivateCapabilities::DEBUG_FNS) { + let name = program.0.get(); + unsafe { gl.object_label(glow::PROGRAM, name, Some(label)) }; + } + } + + let mut name_binding_map = NameBindingMap::default(); + let mut immediates_items = ArrayVec::<_, { crate::MAX_CONCURRENT_SHADER_STAGES }>::new(); + let mut sampler_map = [None; super::MAX_TEXTURE_SLOTS]; + let mut has_stages = wgt::ShaderStages::empty(); + let mut shaders_to_delete = ArrayVec::<_, { crate::MAX_CONCURRENT_SHADER_STAGES }>::new(); + let mut clip_distance_count = 0; + + for &(naga_stage, stage) in &shaders { + has_stages |= map_naga_stage(naga_stage); + let pc_item = { + immediates_items.push(Vec::new()); + immediates_items.last_mut().unwrap() + }; + let context = CompilationContext { + layout, + sampler_map: &mut sampler_map, + name_binding_map: &mut name_binding_map, + immediates_items: pc_item, + multiview_mask, + clip_distance_count: &mut clip_distance_count, + }; + + let shader = Self::create_shader(gl, naga_stage, stage, context, program)?; + shaders_to_delete.push(shader); + } + + // Create empty fragment shader if only vertex shader is present + if has_stages == wgt::ShaderStages::VERTEX { + let shader_src = format!("#version {glsl_version}\n void main(void) {{}}",); + log::debug!("Only vertex shader is present. Creating an empty fragment shader",); + let shader = unsafe { + Self::compile_shader( + gl, + &shader_src, + naga::ShaderStage::Fragment, + Some("(wgpu internal) dummy fragment shader"), + ) + }?; + shaders_to_delete.push(shader); + } + + for &shader in shaders_to_delete.iter() { + unsafe { gl.attach_shader(program, shader) }; + } + unsafe { gl.link_program(program) }; + + for shader in shaders_to_delete { + unsafe { gl.delete_shader(shader) }; + } + + log::debug!("\tLinked program {program:?}"); + + let linked_ok = unsafe { gl.get_program_link_status(program) }; + let msg = unsafe { gl.get_program_info_log(program) }; + if !linked_ok { + return Err(crate::PipelineError::Linkage(has_stages, msg)); + } + if !msg.is_empty() { + log::debug!("\tLink message: {msg}"); + } + + if !private_caps.contains(PrivateCapabilities::SHADER_BINDING_LAYOUT) { + // This remapping is only needed if we aren't able to put the binding layout + // in the shader. We can't remap storage buffers this way. + unsafe { gl.use_program(Some(program)) }; + for (ref name, (register, slot)) in name_binding_map { + log::trace!("Get binding {name:?} from program {program:?}"); + match register { + super::BindingRegister::UniformBuffers => { + let index = unsafe { gl.get_uniform_block_index(program, name) }.unwrap(); + log::trace!("\tBinding slot {slot} to block index {index}"); + unsafe { gl.uniform_block_binding(program, index, slot as _) }; + } + super::BindingRegister::StorageBuffers => { + let index = + unsafe { gl.get_shader_storage_block_index(program, name) }.unwrap(); + log::error!("Unable to re-map shader storage block {name} to {index}"); + return Err(crate::DeviceError::Lost.into()); + } + super::BindingRegister::Textures | super::BindingRegister::Images => { + let location = unsafe { gl.get_uniform_location(program, name) }; + unsafe { gl.uniform_1_i32(location.as_ref(), slot as _) }; + } + } + } + } + + let mut uniforms = ArrayVec::new(); + + for (stage_idx, stage_items) in immediates_items.into_iter().enumerate() { + for item in stage_items { + let naga_module = &shaders[stage_idx].1.module.source.module; + let type_inner = &naga_module.types[item.ty].inner; + + let location = unsafe { gl.get_uniform_location(program, &item.access_path) }; + + log::trace!( + "immediate data item: name={}, ty={:?}, offset={}, location={:?}", + item.access_path, + type_inner, + item.offset, + location, + ); + + if let Some(location) = location { + uniforms.push(super::ImmediateDesc { + location, + offset: item.offset, + size_bytes: type_inner.size(naga_module.to_ctx()), + ty: type_inner.clone(), + }); + } + } + } + + let first_instance_location = if has_stages.contains(wgt::ShaderStages::VERTEX) { + // If this returns none (the uniform isn't active), that's fine, we just won't set it. + unsafe { gl.get_uniform_location(program, naga::back::glsl::FIRST_INSTANCE_BINDING) } + } else { + None + }; + + Ok(Arc::new(super::PipelineInner { + program, + sampler_map, + first_instance_location, + immediates_descs: uniforms, + clip_distance_count, + })) + } +} + +impl crate::Device for super::Device { + type A = super::Api; + + unsafe fn create_buffer( + &self, + desc: &crate::BufferDescriptor, + ) -> Result { + let target = if desc.usage.contains(wgt::BufferUses::INDEX) { + glow::ELEMENT_ARRAY_BUFFER + } else { + glow::ARRAY_BUFFER + }; + + let emulate_map = self + .shared + .workarounds + .contains(super::Workarounds::EMULATE_BUFFER_MAP) + || !self + .shared + .private_caps + .contains(PrivateCapabilities::BUFFER_ALLOCATION); + + if emulate_map && desc.usage.intersects(wgt::BufferUses::MAP_WRITE) { + return Ok(super::Buffer { + raw: None, + target, + size: desc.size, + map_flags: 0, + data: Some(Arc::new(MaybeMutex::new(vec![0; desc.size as usize]))), + offset_of_current_mapping: Arc::new(MaybeMutex::new(0)), + }); + } + + let gl = &self.shared.context.lock(); + + let target = if desc.usage.contains(wgt::BufferUses::INDEX) { + glow::ELEMENT_ARRAY_BUFFER + } else { + glow::ARRAY_BUFFER + }; + + let is_host_visible = desc + .usage + .intersects(wgt::BufferUses::MAP_READ | wgt::BufferUses::MAP_WRITE); + let is_coherent = desc + .memory_flags + .contains(crate::MemoryFlags::PREFER_COHERENT); + + let mut map_flags = 0; + if desc.usage.contains(wgt::BufferUses::MAP_READ) { + map_flags |= glow::MAP_READ_BIT; + } + if desc.usage.contains(wgt::BufferUses::MAP_WRITE) { + map_flags |= glow::MAP_WRITE_BIT; + } + + let raw = Some(unsafe { gl.create_buffer() }.map_err(|_| crate::DeviceError::OutOfMemory)?); + unsafe { gl.bind_buffer(target, raw) }; + let raw_size = desc + .size + .try_into() + .map_err(|_| crate::DeviceError::OutOfMemory)?; + + if self + .shared + .private_caps + .contains(PrivateCapabilities::BUFFER_ALLOCATION) + { + if is_host_visible { + map_flags |= glow::MAP_PERSISTENT_BIT; + if is_coherent { + map_flags |= glow::MAP_COHERENT_BIT; + } + } + // TODO: may also be required for other calls involving `buffer_sub_data_u8_slice` (e.g. copy buffer to buffer and clear buffer) + if desc.usage.intersects(wgt::BufferUses::QUERY_RESOLVE) { + map_flags |= glow::DYNAMIC_STORAGE_BIT; + } + unsafe { gl.buffer_storage(target, raw_size, None, map_flags) }; + } else { + assert!(!is_coherent); + let usage = if is_host_visible { + if desc.usage.contains(wgt::BufferUses::MAP_READ) { + glow::STREAM_READ + } else { + glow::DYNAMIC_DRAW + } + } else { + // Even if the usage doesn't contain SRC_READ, we update it internally at least once + // Some vendors take usage very literally and STATIC_DRAW will freeze us with an empty buffer + // https://github.com/gfx-rs/wgpu/issues/3371 + glow::DYNAMIC_DRAW + }; + unsafe { gl.buffer_data_size(target, raw_size, usage) }; + } + + unsafe { gl.bind_buffer(target, None) }; + + if !is_coherent && desc.usage.contains(wgt::BufferUses::MAP_WRITE) { + map_flags |= glow::MAP_FLUSH_EXPLICIT_BIT; + } + //TODO: do we need `glow::MAP_UNSYNCHRONIZED_BIT`? + + #[cfg(native)] + if let Some(label) = desc.label { + if self + .shared + .private_caps + .contains(PrivateCapabilities::DEBUG_FNS) + { + let name = raw.map_or(0, |buf| buf.0.get()); + unsafe { gl.object_label(glow::BUFFER, name, Some(label)) }; + } + } + + let data = if emulate_map && desc.usage.contains(wgt::BufferUses::MAP_READ) { + Some(Arc::new(MaybeMutex::new(vec![0; desc.size as usize]))) + } else { + None + }; + + self.counters.buffers.add(1); + + Ok(super::Buffer { + raw, + target, + size: desc.size, + map_flags, + data, + offset_of_current_mapping: Arc::new(MaybeMutex::new(0)), + }) + } + + unsafe fn destroy_buffer(&self, buffer: super::Buffer) { + if let Some(raw) = buffer.raw { + let gl = &self.shared.context.lock(); + unsafe { gl.delete_buffer(raw) }; + } + + self.counters.buffers.sub(1); + } + + unsafe fn add_raw_buffer(&self, _buffer: &super::Buffer) { + self.counters.buffers.add(1); + } + + unsafe fn map_buffer( + &self, + buffer: &super::Buffer, + range: crate::MemoryRange, + ) -> Result { + let is_coherent = buffer.map_flags & glow::MAP_COHERENT_BIT != 0; + let ptr = match buffer.raw { + None => { + let mut vec = lock(buffer.data.as_ref().unwrap()); + let slice = &mut vec.as_mut_slice()[range.start as usize..range.end as usize]; + slice.as_mut_ptr() + } + Some(raw) => { + let gl = &self.shared.context.lock(); + unsafe { gl.bind_buffer(buffer.target, Some(raw)) }; + let ptr = if let Some(ref map_read_allocation) = buffer.data { + let mut guard = lock(map_read_allocation); + let slice = guard.as_mut_slice(); + unsafe { self.shared.get_buffer_sub_data(gl, buffer.target, 0, slice) }; + slice.as_mut_ptr() + } else { + *lock(&buffer.offset_of_current_mapping) = range.start; + unsafe { + gl.map_buffer_range( + buffer.target, + range.start as i32, + (range.end - range.start) as i32, + buffer.map_flags, + ) + } + }; + unsafe { gl.bind_buffer(buffer.target, None) }; + ptr + } + }; + Ok(crate::BufferMapping { + ptr: ptr::NonNull::new(ptr).ok_or(crate::DeviceError::Lost)?, + is_coherent, + }) + } + unsafe fn unmap_buffer(&self, buffer: &super::Buffer) { + if let Some(raw) = buffer.raw { + if buffer.data.is_none() { + let gl = &self.shared.context.lock(); + unsafe { gl.bind_buffer(buffer.target, Some(raw)) }; + unsafe { gl.unmap_buffer(buffer.target) }; + unsafe { gl.bind_buffer(buffer.target, None) }; + *lock(&buffer.offset_of_current_mapping) = 0; + } + } + } + unsafe fn flush_mapped_ranges(&self, buffer: &super::Buffer, ranges: I) + where + I: Iterator, + { + if let Some(raw) = buffer.raw { + if buffer.data.is_none() { + let gl = &self.shared.context.lock(); + unsafe { gl.bind_buffer(buffer.target, Some(raw)) }; + for range in ranges { + let offset_of_current_mapping = *lock(&buffer.offset_of_current_mapping); + unsafe { + gl.flush_mapped_buffer_range( + buffer.target, + (range.start - offset_of_current_mapping) as i32, + (range.end - range.start) as i32, + ) + }; + } + } + } + } + unsafe fn invalidate_mapped_ranges(&self, _buffer: &super::Buffer, _ranges: I) { + //TODO: do we need to do anything? + } + + unsafe fn create_texture( + &self, + desc: &crate::TextureDescriptor, + ) -> Result { + let gl = &self.shared.context.lock(); + + let render_usage = wgt::TextureUses::COLOR_TARGET + | wgt::TextureUses::DEPTH_STENCIL_WRITE + | wgt::TextureUses::DEPTH_STENCIL_READ + | wgt::TextureUses::TRANSIENT; + let format_desc = self.shared.describe_texture_format(desc.format); + + let inner = if render_usage.contains(desc.usage) + && desc.dimension == wgt::TextureDimension::D2 + && desc.size.depth_or_array_layers == 1 + { + let raw = unsafe { gl.create_renderbuffer().unwrap() }; + unsafe { gl.bind_renderbuffer(glow::RENDERBUFFER, Some(raw)) }; + if desc.sample_count > 1 { + unsafe { + gl.renderbuffer_storage_multisample( + glow::RENDERBUFFER, + desc.sample_count as i32, + format_desc.internal, + desc.size.width as i32, + desc.size.height as i32, + ) + }; + } else { + unsafe { + gl.renderbuffer_storage( + glow::RENDERBUFFER, + format_desc.internal, + desc.size.width as i32, + desc.size.height as i32, + ) + }; + } + + #[cfg(native)] + if let Some(label) = desc.label { + if self + .shared + .private_caps + .contains(PrivateCapabilities::DEBUG_FNS) + { + let name = raw.0.get(); + unsafe { gl.object_label(glow::RENDERBUFFER, name, Some(label)) }; + } + } + + unsafe { gl.bind_renderbuffer(glow::RENDERBUFFER, None) }; + super::TextureInner::Renderbuffer { raw } + } else { + let raw = unsafe { gl.create_texture().unwrap() }; + let target = super::Texture::get_info_from_desc(desc); + + unsafe { gl.bind_texture(target, Some(raw)) }; + //Note: this has to be done before defining the storage! + match desc.format.sample_type(None, Some(self.shared.features)) { + Some( + wgt::TextureSampleType::Float { filterable: false } + | wgt::TextureSampleType::Uint + | wgt::TextureSampleType::Sint, + ) => { + // reset default filtering mode + unsafe { + gl.tex_parameter_i32(target, glow::TEXTURE_MIN_FILTER, glow::NEAREST as i32) + }; + unsafe { + gl.tex_parameter_i32(target, glow::TEXTURE_MAG_FILTER, glow::NEAREST as i32) + }; + } + _ => {} + } + + if conv::is_layered_target(target) { + unsafe { + if self + .shared + .private_caps + .contains(PrivateCapabilities::TEXTURE_STORAGE) + { + gl.tex_storage_3d( + target, + desc.mip_level_count as i32, + format_desc.internal, + desc.size.width as i32, + desc.size.height as i32, + desc.size.depth_or_array_layers as i32, + ) + } else if target == glow::TEXTURE_3D { + let mut width = desc.size.width; + let mut height = desc.size.height; + let mut depth = desc.size.depth_or_array_layers; + for i in 0..desc.mip_level_count { + gl.tex_image_3d( + target, + i as i32, + format_desc.internal as i32, + width as i32, + height as i32, + depth as i32, + 0, + format_desc.external, + format_desc.data_type, + glow::PixelUnpackData::Slice(None), + ); + width = max(1, width / 2); + height = max(1, height / 2); + depth = max(1, depth / 2); + } + } else { + let mut width = desc.size.width; + let mut height = desc.size.height; + for i in 0..desc.mip_level_count { + gl.tex_image_3d( + target, + i as i32, + format_desc.internal as i32, + width as i32, + height as i32, + desc.size.depth_or_array_layers as i32, + 0, + format_desc.external, + format_desc.data_type, + glow::PixelUnpackData::Slice(None), + ); + width = max(1, width / 2); + height = max(1, height / 2); + } + } + }; + } else if desc.sample_count > 1 { + unsafe { + gl.tex_storage_2d_multisample( + target, + desc.sample_count as i32, + format_desc.internal, + desc.size.width as i32, + desc.size.height as i32, + true, + ) + }; + } else { + unsafe { + if self + .shared + .private_caps + .contains(PrivateCapabilities::TEXTURE_STORAGE) + { + gl.tex_storage_2d( + target, + desc.mip_level_count as i32, + format_desc.internal, + desc.size.width as i32, + desc.size.height as i32, + ) + } else if target == glow::TEXTURE_CUBE_MAP { + let mut width = desc.size.width; + let mut height = desc.size.height; + for i in 0..desc.mip_level_count { + for face in [ + glow::TEXTURE_CUBE_MAP_POSITIVE_X, + glow::TEXTURE_CUBE_MAP_NEGATIVE_X, + glow::TEXTURE_CUBE_MAP_POSITIVE_Y, + glow::TEXTURE_CUBE_MAP_NEGATIVE_Y, + glow::TEXTURE_CUBE_MAP_POSITIVE_Z, + glow::TEXTURE_CUBE_MAP_NEGATIVE_Z, + ] { + gl.tex_image_2d( + face, + i as i32, + format_desc.internal as i32, + width as i32, + height as i32, + 0, + format_desc.external, + format_desc.data_type, + glow::PixelUnpackData::Slice(None), + ); + } + width = max(1, width / 2); + height = max(1, height / 2); + } + } else { + let mut width = desc.size.width; + let mut height = desc.size.height; + for i in 0..desc.mip_level_count { + gl.tex_image_2d( + target, + i as i32, + format_desc.internal as i32, + width as i32, + height as i32, + 0, + format_desc.external, + format_desc.data_type, + glow::PixelUnpackData::Slice(None), + ); + width = max(1, width / 2); + height = max(1, height / 2); + } + } + }; + } + + #[cfg(native)] + if let Some(label) = desc.label { + if self + .shared + .private_caps + .contains(PrivateCapabilities::DEBUG_FNS) + { + let name = raw.0.get(); + unsafe { gl.object_label(glow::TEXTURE, name, Some(label)) }; + } + } + + unsafe { gl.bind_texture(target, None) }; + super::TextureInner::Texture { raw, target } + }; + + self.counters.textures.add(1); + + Ok(super::Texture { + inner, + drop_guard: None, + mip_level_count: desc.mip_level_count, + array_layer_count: desc.array_layer_count(), + format: desc.format, + format_desc, + copy_size: desc.copy_extent(), + }) + } + + unsafe fn destroy_texture(&self, texture: super::Texture) { + if texture.drop_guard.is_none() { + let gl = &self.shared.context.lock(); + match texture.inner { + super::TextureInner::Renderbuffer { raw, .. } => { + unsafe { gl.delete_renderbuffer(raw) }; + } + super::TextureInner::DefaultRenderbuffer => {} + super::TextureInner::Texture { raw, .. } => { + unsafe { gl.delete_texture(raw) }; + } + #[cfg(webgl)] + super::TextureInner::ExternalFramebuffer { .. } => {} + #[cfg(native)] + super::TextureInner::ExternalNativeFramebuffer { .. } => {} + } + } + + // For clarity, we explicitly drop the drop guard. Although this has no real semantic effect as the + // end of the scope will drop the drop guard since this function takes ownership of the texture. + drop(texture.drop_guard); + + self.counters.textures.sub(1); + } + + unsafe fn add_raw_texture(&self, _texture: &super::Texture) { + self.counters.textures.add(1); + } + + unsafe fn create_texture_view( + &self, + texture: &super::Texture, + desc: &crate::TextureViewDescriptor, + ) -> Result { + self.counters.texture_views.add(1); + Ok(super::TextureView { + //TODO: use `conv::map_view_dimension(desc.dimension)`? + inner: texture.inner.clone(), + aspects: crate::FormatAspects::new(texture.format, desc.range.aspect), + mip_levels: desc.range.mip_range(texture.mip_level_count), + array_layers: desc.range.layer_range(texture.array_layer_count), + format: texture.format, + }) + } + + unsafe fn destroy_texture_view(&self, _view: super::TextureView) { + self.counters.texture_views.sub(1); + } + + unsafe fn create_sampler( + &self, + desc: &crate::SamplerDescriptor, + ) -> Result { + let gl = &self.shared.context.lock(); + + let raw = unsafe { gl.create_sampler().unwrap() }; + + let (min, mag) = + conv::map_filter_modes(desc.min_filter, desc.mag_filter, desc.mipmap_filter); + + unsafe { gl.sampler_parameter_i32(raw, glow::TEXTURE_MIN_FILTER, min as i32) }; + unsafe { gl.sampler_parameter_i32(raw, glow::TEXTURE_MAG_FILTER, mag as i32) }; + + unsafe { + gl.sampler_parameter_i32( + raw, + glow::TEXTURE_WRAP_S, + conv::map_address_mode(desc.address_modes[0]) as i32, + ) + }; + unsafe { + gl.sampler_parameter_i32( + raw, + glow::TEXTURE_WRAP_T, + conv::map_address_mode(desc.address_modes[1]) as i32, + ) + }; + unsafe { + gl.sampler_parameter_i32( + raw, + glow::TEXTURE_WRAP_R, + conv::map_address_mode(desc.address_modes[2]) as i32, + ) + }; + + if let Some(border_color) = desc.border_color { + let border = match border_color { + wgt::SamplerBorderColor::TransparentBlack | wgt::SamplerBorderColor::Zero => { + [0.0; 4] + } + wgt::SamplerBorderColor::OpaqueBlack => [0.0, 0.0, 0.0, 1.0], + wgt::SamplerBorderColor::OpaqueWhite => [1.0; 4], + }; + unsafe { gl.sampler_parameter_f32_slice(raw, glow::TEXTURE_BORDER_COLOR, &border) }; + } + + unsafe { gl.sampler_parameter_f32(raw, glow::TEXTURE_MIN_LOD, desc.lod_clamp.start) }; + unsafe { gl.sampler_parameter_f32(raw, glow::TEXTURE_MAX_LOD, desc.lod_clamp.end) }; + + // If clamp is not 1, we know anisotropy is supported up to 16x + if desc.anisotropy_clamp != 1 { + unsafe { + gl.sampler_parameter_i32( + raw, + glow::TEXTURE_MAX_ANISOTROPY, + desc.anisotropy_clamp as i32, + ) + }; + } + + //set_param_float(glow::TEXTURE_LOD_BIAS, info.lod_bias.0); + + if let Some(compare) = desc.compare { + unsafe { + gl.sampler_parameter_i32( + raw, + glow::TEXTURE_COMPARE_MODE, + glow::COMPARE_REF_TO_TEXTURE as i32, + ) + }; + unsafe { + gl.sampler_parameter_i32( + raw, + glow::TEXTURE_COMPARE_FUNC, + conv::map_compare_func(compare) as i32, + ) + }; + } + + #[cfg(native)] + if let Some(label) = desc.label { + if self + .shared + .private_caps + .contains(PrivateCapabilities::DEBUG_FNS) + { + let name = raw.0.get(); + unsafe { gl.object_label(glow::SAMPLER, name, Some(label)) }; + } + } + + self.counters.samplers.add(1); + + Ok(super::Sampler { raw }) + } + + unsafe fn destroy_sampler(&self, sampler: super::Sampler) { + let gl = &self.shared.context.lock(); + unsafe { gl.delete_sampler(sampler.raw) }; + self.counters.samplers.sub(1); + } + + unsafe fn create_command_encoder( + &self, + _desc: &crate::CommandEncoderDescriptor, + ) -> Result { + self.counters.command_encoders.add(1); + + Ok(super::CommandEncoder { + cmd_buffer: super::CommandBuffer::default(), + state: Default::default(), + private_caps: self.shared.private_caps, + counters: Arc::clone(&self.counters), + }) + } + + unsafe fn create_bind_group_layout( + &self, + desc: &crate::BindGroupLayoutDescriptor, + ) -> Result { + self.counters.bind_group_layouts.add(1); + Ok(super::BindGroupLayout { + entries: Arc::from(desc.entries), + }) + } + + unsafe fn destroy_bind_group_layout(&self, _bg_layout: super::BindGroupLayout) { + self.counters.bind_group_layouts.sub(1); + } + + unsafe fn create_pipeline_layout( + &self, + desc: &crate::PipelineLayoutDescriptor, + ) -> Result { + use naga::back::glsl; + + let mut group_infos = Vec::with_capacity(desc.bind_group_layouts.len()); + let mut num_samplers = 0u8; + let mut num_textures = 0u8; + let mut num_images = 0u8; + let mut num_uniform_buffers = 0u8; + let mut num_storage_buffers = 0u8; + + let mut writer_flags = glsl::WriterFlags::ADJUST_COORDINATE_SPACE; + writer_flags.set( + glsl::WriterFlags::TEXTURE_SHADOW_LOD, + self.shared + .private_caps + .contains(PrivateCapabilities::SHADER_TEXTURE_SHADOW_LOD), + ); + writer_flags.set( + glsl::WriterFlags::DRAW_PARAMETERS, + self.shared + .private_caps + .contains(PrivateCapabilities::FULLY_FEATURED_INSTANCING), + ); + // We always force point size to be written and it will be ignored by the driver if it's not a point list primitive. + // https://github.com/gfx-rs/wgpu/pull/3440/files#r1095726950 + writer_flags.set(glsl::WriterFlags::FORCE_POINT_SIZE, true); + let mut binding_map = glsl::BindingMap::default(); + + for (group_index, bg_layout) in desc.bind_group_layouts.iter().enumerate() { + let Some(bg_layout) = bg_layout else { + group_infos.push(None); + continue; + }; + + // create a vector with the size enough to hold all the bindings, filled with `!0` + let mut binding_to_slot = vec![ + !0; + bg_layout + .entries + .iter() + .map(|b| b.binding) + .max() + .map_or(0, |idx| idx as usize + 1) + ] + .into_boxed_slice(); + + for entry in bg_layout.entries.iter() { + let counter = match entry.ty { + wgt::BindingType::Sampler { .. } => &mut num_samplers, + wgt::BindingType::Texture { .. } => &mut num_textures, + wgt::BindingType::StorageTexture { .. } => &mut num_images, + wgt::BindingType::Buffer { + ty: wgt::BufferBindingType::Uniform, + .. + } => &mut num_uniform_buffers, + wgt::BindingType::Buffer { + ty: wgt::BufferBindingType::Storage { .. }, + .. + } => &mut num_storage_buffers, + wgt::BindingType::AccelerationStructure { .. } => unimplemented!(), + wgt::BindingType::ExternalTexture => unimplemented!(), + }; + + binding_to_slot[entry.binding as usize] = *counter; + let br = naga::ResourceBinding { + group: group_index as u32, + binding: entry.binding, + }; + binding_map.insert(br, *counter); + *counter += entry.count.map_or(1, |c| c.get() as u8); + } + + group_infos.push(Some(super::BindGroupLayoutInfo { + entries: Arc::clone(&bg_layout.entries), + binding_to_slot, + })); + } + + self.counters.pipeline_layouts.add(1); + + Ok(super::PipelineLayout { + group_infos: group_infos.into_boxed_slice(), + naga_options: glsl::Options { + version: self.shared.shading_language_version, + writer_flags, + binding_map, + zero_initialize_workgroup_memory: true, + }, + }) + } + + unsafe fn destroy_pipeline_layout(&self, _pipeline_layout: super::PipelineLayout) { + self.counters.pipeline_layouts.sub(1); + } + + unsafe fn create_bind_group( + &self, + desc: &crate::BindGroupDescriptor< + super::BindGroupLayout, + super::Buffer, + super::Sampler, + super::TextureView, + super::AccelerationStructure, + >, + ) -> Result { + let mut contents = Vec::new(); + + let layout_and_entry_iter = desc.entries.iter().map(|entry| { + let layout = desc + .layout + .entries + .iter() + .find(|layout_entry| layout_entry.binding == entry.binding) + .expect("internal error: no layout entry found with binding slot"); + (entry, layout) + }); + for (entry, layout) in layout_and_entry_iter { + let binding = match layout.ty { + wgt::BindingType::Buffer { .. } => { + let bb = &desc.buffers[entry.resource_index as usize]; + super::RawBinding::Buffer { + raw: bb.buffer.raw.unwrap(), + offset: bb.offset as i32, + size: match bb.size { + Some(s) => s.get() as i32, + None => (bb.buffer.size - bb.offset) as i32, + }, + } + } + wgt::BindingType::Sampler { .. } => { + let sampler = desc.samplers[entry.resource_index as usize]; + super::RawBinding::Sampler(sampler.raw) + } + wgt::BindingType::Texture { view_dimension, .. } => { + let view = desc.textures[entry.resource_index as usize].view; + if view.array_layers.start != 0 { + log::error!("Unable to create a sampled texture binding for non-zero array layer.\n{}", + "This is an implementation problem of wgpu-hal/gles backend.") + } + let (raw, target) = view.inner.as_native(); + + super::Texture::log_failing_target_heuristics(view_dimension, target); + + super::RawBinding::Texture { + raw, + target, + aspects: view.aspects, + mip_levels: view.mip_levels.clone(), + } + } + wgt::BindingType::StorageTexture { + access, + format, + view_dimension, + } => { + let view = desc.textures[entry.resource_index as usize].view; + let format_desc = self.shared.describe_texture_format(format); + let (raw, _target) = view.inner.as_native(); + super::RawBinding::Image(super::ImageBinding { + raw, + mip_level: view.mip_levels.start, + array_layer: match view_dimension { + wgt::TextureViewDimension::D2Array + | wgt::TextureViewDimension::CubeArray => None, + _ => Some(view.array_layers.start), + }, + access: conv::map_storage_access(access), + format: format_desc.internal, + }) + } + wgt::BindingType::AccelerationStructure { .. } => unimplemented!(), + wgt::BindingType::ExternalTexture => unimplemented!(), + }; + contents.push(binding); + } + + self.counters.bind_groups.add(1); + + Ok(super::BindGroup { + contents: contents.into_boxed_slice(), + }) + } + + unsafe fn destroy_bind_group(&self, _group: super::BindGroup) { + self.counters.bind_groups.sub(1); + } + + unsafe fn create_shader_module( + &self, + desc: &crate::ShaderModuleDescriptor, + shader: crate::ShaderInput, + ) -> Result { + self.counters.shader_modules.add(1); + + Ok(super::ShaderModule { + source: match shader { + crate::ShaderInput::Naga(naga) => naga, + // The backend doesn't yet expose this feature so it should be fine + crate::ShaderInput::Glsl { .. } => unimplemented!(), + crate::ShaderInput::SpirV(_) + | crate::ShaderInput::MetalLib { .. } + | crate::ShaderInput::Msl { .. } + | crate::ShaderInput::Dxil { .. } + | crate::ShaderInput::Hlsl { .. } => { + unreachable!() + } + }, + label: desc.label.map(|str| str.to_string()), + id: self.shared.next_shader_id.fetch_add(1, Ordering::Relaxed), + }) + } + + unsafe fn destroy_shader_module(&self, _module: super::ShaderModule) { + self.counters.shader_modules.sub(1); + } + + unsafe fn create_render_pipeline( + &self, + desc: &crate::RenderPipelineDescriptor< + super::PipelineLayout, + super::ShaderModule, + super::PipelineCache, + >, + ) -> Result { + let (vertex_stage, vertex_buffers) = match &desc.vertex_processor { + crate::VertexProcessor::Standard { + vertex_buffers, + ref vertex_stage, + } => (vertex_stage, vertex_buffers), + crate::VertexProcessor::Mesh { .. } => unreachable!(), + }; + let gl = &self.shared.context.lock(); + let mut shaders = ArrayVec::new(); + shaders.push((naga::ShaderStage::Vertex, vertex_stage)); + if let Some(ref fs) = desc.fragment_stage { + shaders.push((naga::ShaderStage::Fragment, fs)); + } + let inner = unsafe { + self.create_pipeline(gl, shaders, desc.layout, desc.label, desc.multiview_mask) + }?; + + let (vertex_buffers, vertex_attributes) = { + let mut buffers = Vec::new(); + let mut attributes = Vec::new(); + for (index, vb_layout) in vertex_buffers.iter().enumerate() { + buffers.push(super::VertexBufferDesc { + step: vb_layout.step_mode, + stride: vb_layout.array_stride as u32, + }); + for vat in vb_layout.attributes.iter() { + let format_desc = conv::describe_vertex_format(vat.format); + attributes.push(super::AttributeDesc { + location: vat.shader_location, + offset: vat.offset as u32, + buffer_index: index as u32, + format_desc, + }); + } + } + (buffers.into_boxed_slice(), attributes.into_boxed_slice()) + }; + + let color_targets = { + let mut targets = Vec::new(); + for ct in desc.color_targets.iter().filter_map(|at| at.as_ref()) { + targets.push(super::ColorTargetDesc { + mask: ct.write_mask, + blend: ct.blend.as_ref().map(conv::map_blend), + }); + } + //Note: if any of the states are different, and `INDEPENDENT_BLEND` flag + // is not exposed, then this pipeline will not bind correctly. + targets.into_boxed_slice() + }; + + self.counters.render_pipelines.add(1); + + Ok(super::RenderPipeline { + inner, + primitive: desc.primitive, + vertex_buffers, + vertex_attributes, + color_targets, + depth: desc.depth_stencil.as_ref().map(|ds| super::DepthState { + function: conv::map_compare_func(ds.depth_compare.unwrap_or_default()), + mask: ds.depth_write_enabled.unwrap_or_default(), + }), + depth_bias: desc + .depth_stencil + .as_ref() + .map(|ds| ds.bias) + .unwrap_or_default(), + stencil: desc + .depth_stencil + .as_ref() + .map(|ds| conv::map_stencil(&ds.stencil)), + alpha_to_coverage_enabled: desc.multisample.alpha_to_coverage_enabled, + }) + } + + unsafe fn destroy_render_pipeline(&self, pipeline: super::RenderPipeline) { + // If the pipeline only has 2 strong references remaining, they're `pipeline` and `program_cache` + // This is safe to assume as long as: + // - `RenderPipeline` can't be cloned + // - The only place that we can get a new reference is during `program_cache.lock()` + if Arc::strong_count(&pipeline.inner) == 2 { + let gl = &self.shared.context.lock(); + let mut program_cache = self.shared.program_cache.lock(); + program_cache.retain(|_, v| match *v { + Ok(ref p) => p.program != pipeline.inner.program, + Err(_) => false, + }); + unsafe { gl.delete_program(pipeline.inner.program) }; + } + + self.counters.render_pipelines.sub(1); + } + + unsafe fn create_compute_pipeline( + &self, + desc: &crate::ComputePipelineDescriptor< + super::PipelineLayout, + super::ShaderModule, + super::PipelineCache, + >, + ) -> Result { + let gl = &self.shared.context.lock(); + let mut shaders = ArrayVec::new(); + shaders.push((naga::ShaderStage::Compute, &desc.stage)); + let inner = unsafe { self.create_pipeline(gl, shaders, desc.layout, desc.label, None) }?; + + self.counters.compute_pipelines.add(1); + + Ok(super::ComputePipeline { inner }) + } + + unsafe fn destroy_compute_pipeline(&self, pipeline: super::ComputePipeline) { + // If the pipeline only has 2 strong references remaining, they're `pipeline` and `program_cache`` + // This is safe to assume as long as: + // - `ComputePipeline` can't be cloned + // - The only place that we can get a new reference is during `program_cache.lock()` + if Arc::strong_count(&pipeline.inner) == 2 { + let gl = &self.shared.context.lock(); + let mut program_cache = self.shared.program_cache.lock(); + program_cache.retain(|_, v| match *v { + Ok(ref p) => p.program != pipeline.inner.program, + Err(_) => false, + }); + unsafe { gl.delete_program(pipeline.inner.program) }; + } + + self.counters.compute_pipelines.sub(1); + } + + unsafe fn create_pipeline_cache( + &self, + _: &crate::PipelineCacheDescriptor<'_>, + ) -> Result { + // Even though the cache doesn't do anything, we still return something here + // as the least bad option + Ok(super::PipelineCache) + } + unsafe fn destroy_pipeline_cache(&self, _: super::PipelineCache) {} + + #[cfg_attr(target_arch = "wasm32", allow(unused))] + unsafe fn create_query_set( + &self, + desc: &wgt::QuerySetDescriptor, + ) -> Result { + let gl = &self.shared.context.lock(); + + let mut queries = Vec::with_capacity(desc.count as usize); + for _ in 0..desc.count { + let query = + unsafe { gl.create_query() }.map_err(|_| crate::DeviceError::OutOfMemory)?; + + // We aren't really able to, in general, label queries. + // + // We could take a timestamp here to "initialize" the query, + // but that's a bit of a hack, and we don't want to insert + // random timestamps into the command stream of we don't have to. + + queries.push(query); + } + + self.counters.query_sets.add(1); + + Ok(super::QuerySet { + queries: queries.into_boxed_slice(), + target: match desc.ty { + wgt::QueryType::Occlusion => glow::ANY_SAMPLES_PASSED_CONSERVATIVE, + wgt::QueryType::Timestamp => glow::TIMESTAMP, + _ => unimplemented!(), + }, + }) + } + + unsafe fn destroy_query_set(&self, set: super::QuerySet) { + let gl = &self.shared.context.lock(); + for &query in set.queries.iter() { + unsafe { gl.delete_query(query) }; + } + self.counters.query_sets.sub(1); + } + + unsafe fn create_fence(&self) -> Result { + self.counters.fences.add(1); + Ok(super::Fence::new(&self.shared.options)) + } + + unsafe fn destroy_fence(&self, fence: super::Fence) { + let gl = &self.shared.context.lock(); + fence.destroy(gl); + self.counters.fences.sub(1); + } + + unsafe fn get_fence_value( + &self, + fence: &super::Fence, + ) -> Result { + #[cfg_attr(target_arch = "wasm32", allow(clippy::needless_borrow))] + Ok(fence.get_latest(&self.shared.context.lock())) + } + unsafe fn wait( + &self, + fence: &super::Fence, + wait_value: crate::FenceValue, + timeout: Option, + ) -> Result { + if fence.satisfied(wait_value) { + return Ok(true); + } + + let gl = &self.shared.context.lock(); + // MAX_CLIENT_WAIT_TIMEOUT_WEBGL is: + // - 1s in Gecko https://searchfox.org/mozilla-central/rev/754074e05178e017ef6c3d8e30428ffa8f1b794d/dom/canvas/WebGLTypes.h#1386 + // - 0 in WebKit https://github.com/WebKit/WebKit/blob/4ef90d4672ca50267c0971b85db403d9684508ea/Source/WebCore/html/canvas/WebGL2RenderingContext.cpp#L110 + // - 0 in Chromium https://source.chromium.org/chromium/chromium/src/+/main:third_party/blink/renderer/modules/webgl/webgl2_rendering_context_base.cc;l=112;drc=a3cb0ac4c71ec04abfeaed199e5d63230eca2551 + let timeout_ns = if cfg!(any(webgl, Emscripten)) { + 0 + } else { + timeout + .map(|t| t.as_nanos().min(u32::MAX as u128) as u32) + .unwrap_or(u32::MAX) + }; + fence.wait(gl, wait_value, timeout_ns) + } + + unsafe fn start_graphics_debugger_capture(&self) -> bool { + #[cfg(all(native, feature = "renderdoc"))] + return unsafe { + self.render_doc + .start_frame_capture(self.shared.context.raw_context(), ptr::null_mut()) + }; + #[allow(unreachable_code)] + false + } + unsafe fn stop_graphics_debugger_capture(&self) { + #[cfg(all(native, feature = "renderdoc"))] + unsafe { + self.render_doc + .end_frame_capture(ptr::null_mut(), ptr::null_mut()) + } + } + unsafe fn create_acceleration_structure( + &self, + _desc: &crate::AccelerationStructureDescriptor, + ) -> Result { + unimplemented!() + } + unsafe fn get_acceleration_structure_build_sizes<'a>( + &self, + _desc: &crate::GetAccelerationStructureBuildSizesDescriptor<'a, super::Buffer>, + ) -> crate::AccelerationStructureBuildSizes { + unimplemented!() + } + unsafe fn get_acceleration_structure_device_address( + &self, + _acceleration_structure: &super::AccelerationStructure, + ) -> wgt::BufferAddress { + unimplemented!() + } + unsafe fn destroy_acceleration_structure( + &self, + _acceleration_structure: super::AccelerationStructure, + ) { + } + + fn tlas_instance_to_bytes(&self, _instance: TlasInstance) -> Vec { + unimplemented!() + } + + fn get_internal_counters(&self) -> wgt::HalCounters { + self.counters.as_ref().clone() + } + + fn check_if_oom(&self) -> Result<(), crate::DeviceError> { + Ok(()) + } +} + +#[cfg(send_sync)] +unsafe impl Sync for super::Device {} +#[cfg(send_sync)] +unsafe impl Send for super::Device {} diff --git a/third_party/wgpu-hal-29.0.4/src/gles/egl.rs b/third_party/wgpu-hal-29.0.4/src/gles/egl.rs new file mode 100644 index 0000000..9bcac23 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/egl.rs @@ -0,0 +1,1460 @@ +use alloc::{string::String, sync::Arc, vec::Vec}; +use core::{ffi, mem::ManuallyDrop, ptr, time::Duration}; +use std::sync::LazyLock; + +use glow::HasContext; +use hashbrown::HashMap; +use parking_lot::{MappedMutexGuard, Mutex, MutexGuard, RwLock}; + +/// The amount of time to wait while trying to obtain a lock to the adapter context +const CONTEXT_LOCK_TIMEOUT_SECS: u64 = 6; + +const EGL_CONTEXT_FLAGS_KHR: i32 = 0x30FC; +const EGL_CONTEXT_OPENGL_DEBUG_BIT_KHR: i32 = 0x0001; +const EGL_CONTEXT_OPENGL_ROBUST_ACCESS_EXT: i32 = 0x30BF; +const EGL_PLATFORM_WAYLAND_KHR: u32 = 0x31D8; +const EGL_PLATFORM_X11_KHR: u32 = 0x31D5; +const EGL_PLATFORM_XCB_EXT: u32 = 0x31DC; +const EGL_PLATFORM_XCB_SCREEN_EXT: u32 = 0x31DE; +const EGL_PLATFORM_ANGLE_ANGLE: u32 = 0x3202; +const EGL_PLATFORM_ANGLE_NATIVE_PLATFORM_TYPE_ANGLE: u32 = 0x348F; +const EGL_PLATFORM_ANGLE_DEBUG_LAYERS_ENABLED: u32 = 0x3451; +const EGL_PLATFORM_SURFACELESS_MESA: u32 = 0x31DD; +const EGL_GL_COLORSPACE_KHR: u32 = 0x309D; +const EGL_GL_COLORSPACE_SRGB_KHR: u32 = 0x3089; + +#[cfg(not(Emscripten))] +type EglInstance = khronos_egl::DynamicInstance; + +#[cfg(Emscripten)] +type EglInstance = khronos_egl::Instance; + +type EglLabel = *const ffi::c_void; + +#[allow(clippy::upper_case_acronyms)] +type EGLDEBUGPROCKHR = Option< + unsafe extern "system" fn( + error: khronos_egl::Enum, + command: *const ffi::c_char, + message_type: u32, + thread_label: EglLabel, + object_label: EglLabel, + message: *const ffi::c_char, + ), +>; + +const EGL_DEBUG_MSG_CRITICAL_KHR: u32 = 0x33B9; +const EGL_DEBUG_MSG_ERROR_KHR: u32 = 0x33BA; +const EGL_DEBUG_MSG_WARN_KHR: u32 = 0x33BB; +const EGL_DEBUG_MSG_INFO_KHR: u32 = 0x33BC; + +type EglDebugMessageControlFun = unsafe extern "system" fn( + proc: EGLDEBUGPROCKHR, + attrib_list: *const khronos_egl::Attrib, +) -> ffi::c_int; + +unsafe extern "system" fn egl_debug_proc( + error: khronos_egl::Enum, + command_raw: *const ffi::c_char, + message_type: u32, + _thread_label: EglLabel, + _object_label: EglLabel, + message_raw: *const ffi::c_char, +) { + let log_severity = match message_type { + EGL_DEBUG_MSG_CRITICAL_KHR | EGL_DEBUG_MSG_ERROR_KHR => log::Level::Error, + EGL_DEBUG_MSG_WARN_KHR => log::Level::Warn, + // We intentionally suppress info messages down to debug + // so that users are not inundated with info messages from + // the runtime. + EGL_DEBUG_MSG_INFO_KHR => log::Level::Debug, + _ => log::Level::Trace, + }; + let command = unsafe { ffi::CStr::from_ptr(command_raw) }.to_string_lossy(); + let message = if message_raw.is_null() { + "".into() + } else { + unsafe { ffi::CStr::from_ptr(message_raw) }.to_string_lossy() + }; + + log::log!(log_severity, "EGL '{command}' code 0x{error:x}: {message}",); +} + +#[derive(Clone, Copy, Debug)] +enum SrgbFrameBufferKind { + /// No support for SRGB surface + None, + /// Using EGL 1.5's support for colorspaces + Core, + /// Using EGL_KHR_gl_colorspace + Khr, +} + +/// Choose GLES framebuffer configuration. +fn choose_config( + egl: &EglInstance, + display: khronos_egl::Display, + srgb_kind: SrgbFrameBufferKind, +) -> Result<(khronos_egl::Config, bool), crate::InstanceError> { + //TODO: EGL_SLOW_CONFIG + let tiers = [ + ( + "off-screen", + &[ + khronos_egl::SURFACE_TYPE, + khronos_egl::PBUFFER_BIT, + khronos_egl::RENDERABLE_TYPE, + khronos_egl::OPENGL_ES2_BIT, + ][..], + ), + ( + "presentation", + &[khronos_egl::SURFACE_TYPE, khronos_egl::WINDOW_BIT][..], + ), + #[cfg(not(target_os = "android"))] + ( + "native-render", + &[khronos_egl::NATIVE_RENDERABLE, khronos_egl::TRUE as _][..], + ), + ]; + + let mut attributes = Vec::with_capacity(9); + for tier_max in (0..tiers.len()).rev() { + let name = tiers[tier_max].0; + log::debug!("\tTrying {name}"); + + attributes.clear(); + for &(_, tier_attr) in tiers[..=tier_max].iter() { + attributes.extend_from_slice(tier_attr); + } + // make sure the Alpha is enough to support sRGB + match srgb_kind { + SrgbFrameBufferKind::None => {} + _ => { + attributes.push(khronos_egl::ALPHA_SIZE); + attributes.push(8); + } + } + attributes.push(khronos_egl::NONE); + + match egl.choose_first_config(display, &attributes) { + Ok(Some(config)) => { + if tier_max == 1 { + //Note: this has been confirmed to malfunction on Intel+NV laptops, + // but also on Angle. + log::info!("EGL says it can present to the window but not natively",); + } + // Android emulator can't natively present either. + let tier_threshold = + if cfg!(target_os = "android") || cfg!(windows) || cfg!(target_env = "ohos") { + 1 + } else { + 2 + }; + return Ok((config, tier_max >= tier_threshold)); + } + Ok(None) => { + log::debug!("No config found!"); + } + Err(e) => { + log::error!("error in choose_first_config: {e:?}"); + } + } + } + + // TODO: include diagnostic details that are currently logged + Err(crate::InstanceError::new(String::from( + "unable to find an acceptable EGL framebuffer configuration", + ))) +} + +#[derive(Clone, Debug)] +struct EglContext { + instance: Arc, + version: (i32, i32), + display: khronos_egl::Display, + raw: khronos_egl::Context, + pbuffer: Option, +} + +impl EglContext { + fn make_current(&self) { + self.instance + .make_current(self.display, self.pbuffer, self.pbuffer, Some(self.raw)) + .unwrap(); + } + + fn unmake_current(&self) { + self.instance + .make_current(self.display, None, None, None) + .unwrap(); + } +} + +/// A wrapper around a [`glow::Context`] and the required EGL context that uses locking to guarantee +/// exclusive access when shared with multiple threads. +pub struct AdapterContext { + glow: Mutex>, + egl: Option, +} + +unsafe impl Sync for AdapterContext {} +unsafe impl Send for AdapterContext {} + +impl AdapterContext { + pub fn is_owned(&self) -> bool { + self.egl.is_some() + } + + /// Returns the EGL instance. + /// + /// This provides access to EGL functions and the ability to load GL and EGL extension functions. + pub fn egl_instance(&self) -> Option<&EglInstance> { + self.egl.as_ref().map(|egl| &*egl.instance) + } + + /// Returns the EGLDisplay corresponding to the adapter context. + /// + /// Returns [`None`] if the adapter was externally created. + pub fn raw_display(&self) -> Option<&khronos_egl::Display> { + self.egl.as_ref().map(|egl| &egl.display) + } + + /// Returns the EGL version the adapter context was created with. + /// + /// Returns [`None`] if the adapter was externally created. + pub fn egl_version(&self) -> Option<(i32, i32)> { + self.egl.as_ref().map(|egl| egl.version) + } + + pub fn raw_context(&self) -> *mut ffi::c_void { + match self.egl { + Some(ref egl) => egl.raw.as_ptr(), + None => ptr::null_mut(), + } + } +} + +impl Drop for AdapterContext { + fn drop(&mut self) { + struct CurrentGuard<'a>(&'a EglContext); + impl Drop for CurrentGuard<'_> { + fn drop(&mut self) { + self.0.unmake_current(); + } + } + + // Context must be current when dropped. See safety docs on + // `glow::HasContext`. + // + // NOTE: This is only set to `None` by `Adapter::new_external` which + // requires the context to be current when anything that may be holding + // the `Arc` is dropped. + let _guard = self.egl.as_ref().map(|egl| { + egl.make_current(); + CurrentGuard(egl) + }); + let glow = self.glow.get_mut(); + // SAFETY: Field not used after this. + unsafe { ManuallyDrop::drop(glow) }; + } +} + +struct EglContextLock<'a> { + instance: &'a Arc, + display: khronos_egl::Display, +} + +/// A guard containing a lock to an [`AdapterContext`], while the GL context is kept current. +pub struct AdapterContextLock<'a> { + glow: MutexGuard<'a, ManuallyDrop>, + egl: Option>, +} + +impl<'a> core::ops::Deref for AdapterContextLock<'a> { + type Target = glow::Context; + + fn deref(&self) -> &Self::Target { + &self.glow + } +} + +impl<'a> Drop for AdapterContextLock<'a> { + fn drop(&mut self) { + if let Some(egl) = self.egl.take() { + if let Err(err) = egl.instance.make_current(egl.display, None, None, None) { + log::error!("Failed to make EGL context current: {err:?}"); + } + } + } +} + +impl AdapterContext { + /// Get's the [`glow::Context`] without waiting for a lock + /// + /// # Safety + /// + /// This should only be called when you have manually made sure that the current thread has made + /// the EGL context current and that no other thread also has the EGL context current. + /// Additionally, you must manually make the EGL context **not** current after you are done with + /// it, so that future calls to `lock()` will not fail. + /// + /// > **Note:** Calling this function **will** still lock the [`glow::Context`] which adds an + /// > extra safe-guard against accidental concurrent access to the context. + pub unsafe fn get_without_egl_lock(&self) -> MappedMutexGuard<'_, glow::Context> { + let guard = self + .glow + .try_lock_for(Duration::from_secs(CONTEXT_LOCK_TIMEOUT_SECS)) + .expect("Could not lock adapter context. This is most-likely a deadlock."); + MutexGuard::map(guard, |glow| &mut **glow) + } + + /// Obtain a lock to the EGL context and get handle to the [`glow::Context`] that can be used to + /// do rendering. + #[track_caller] + pub fn lock<'a>(&'a self) -> AdapterContextLock<'a> { + let glow = self + .glow + // Don't lock forever. If it takes longer than 1 second to get the lock we've got a + // deadlock and should panic to show where we got stuck + .try_lock_for(Duration::from_secs(CONTEXT_LOCK_TIMEOUT_SECS)) + .expect("Could not lock adapter context. This is most-likely a deadlock."); + + let egl = self.egl.as_ref().map(|egl| { + egl.make_current(); + EglContextLock { + instance: &egl.instance, + display: egl.display, + } + }); + + AdapterContextLock { glow, egl } + } +} + +#[derive(Debug)] +struct Inner { + /// Note: the context contains a dummy pbuffer (1x1). + /// Required for `eglMakeCurrent` on platforms that doesn't supports `EGL_KHR_surfaceless_context`. + egl: EglContext, + version: (i32, i32), + supports_native_window: bool, + config: khronos_egl::Config, + /// Method by which the framebuffer should support srgb + srgb_kind: SrgbFrameBufferKind, +} + +// Different calls to `eglGetPlatformDisplay` may return the same `Display`, making it a global +// state of all our `EglContext`s. This forces us to track the number of such context to prevent +// terminating the display if it's currently used by another `EglContext`. +static DISPLAYS_REFERENCE_COUNT: LazyLock>> = + LazyLock::new(Default::default); + +fn initialize_display( + egl: &EglInstance, + display: khronos_egl::Display, +) -> Result<(i32, i32), khronos_egl::Error> { + let mut guard = DISPLAYS_REFERENCE_COUNT.lock(); + *guard.entry(display.as_ptr() as usize).or_default() += 1; + + // We don't need to check the reference count here since according to the `eglInitialize` + // documentation, initializing an already initialized EGL display connection has no effect + // besides returning the version numbers. + egl.initialize(display) +} + +fn terminate_display( + egl: &EglInstance, + display: khronos_egl::Display, +) -> Result<(), khronos_egl::Error> { + let key = &(display.as_ptr() as usize); + let mut guard = DISPLAYS_REFERENCE_COUNT.lock(); + let count_ref = guard + .get_mut(key) + .expect("Attempted to decref a display before incref was called"); + + if *count_ref > 1 { + *count_ref -= 1; + + Ok(()) + } else { + guard.remove(key); + + egl.terminate(display) + } +} + +fn instance_err( + message: impl Into, +) -> impl FnOnce(E) -> crate::InstanceError { + move |e| crate::InstanceError::with_source(message.into(), e) +} + +impl Inner { + fn create( + flags: wgt::InstanceFlags, + egl: Arc, + display: khronos_egl::Display, + force_gles_minor_version: wgt::Gles3MinorVersion, + ) -> Result { + let version = initialize_display(&egl, display) + .map_err(instance_err("failed to initialize EGL display connection"))?; + let vendor = egl + .query_string(Some(display), khronos_egl::VENDOR) + .map_err(instance_err("failed to query EGL vendor"))?; + let display_extensions = egl + .query_string(Some(display), khronos_egl::EXTENSIONS) + .map_err(instance_err("failed to query EGL display extensions"))? + .to_string_lossy(); + log::debug!("Display vendor {vendor:?}, version {version:?}",); + log::debug!( + "Display extensions: {:#?}", + display_extensions.split_whitespace().collect::>() + ); + + let srgb_kind = if version >= (1, 5) { + log::debug!("\tEGL surface: +srgb"); + SrgbFrameBufferKind::Core + } else if display_extensions.contains("EGL_KHR_gl_colorspace") { + log::debug!("\tEGL surface: +srgb khr"); + SrgbFrameBufferKind::Khr + } else { + log::debug!("\tEGL surface: -srgb"); + SrgbFrameBufferKind::None + }; + + if log::max_level() >= log::LevelFilter::Trace { + log::trace!("Configurations:"); + let config_count = egl + .get_config_count(display) + .map_err(instance_err("failed to get config count"))?; + let mut configurations = Vec::with_capacity(config_count); + egl.get_configs(display, &mut configurations) + .map_err(instance_err("failed to get configs"))?; + for &config in configurations.iter() { + log::trace!("\tCONFORMANT=0x{:X?}, RENDERABLE=0x{:X?}, NATIVE_RENDERABLE=0x{:X?}, SURFACE_TYPE=0x{:X?}, ALPHA_SIZE={:?}", + egl.get_config_attrib(display, config, khronos_egl::CONFORMANT), + egl.get_config_attrib(display, config, khronos_egl::RENDERABLE_TYPE), + egl.get_config_attrib(display, config, khronos_egl::NATIVE_RENDERABLE), + egl.get_config_attrib(display, config, khronos_egl::SURFACE_TYPE), + egl.get_config_attrib(display, config, khronos_egl::ALPHA_SIZE), + ); + } + } + + let (config, supports_native_window) = choose_config(&egl, display, srgb_kind)?; + + let supports_opengl = if version >= (1, 4) { + let client_apis = egl + .query_string(Some(display), khronos_egl::CLIENT_APIS) + .map_err(instance_err("failed to query EGL client APIs string"))? + .to_string_lossy(); + client_apis + .split(' ') + .any(|client_api| client_api == "OpenGL") + } else { + false + }; + + let mut khr_context_flags = 0; + let supports_khr_context = display_extensions.contains("EGL_KHR_create_context"); + + let mut context_attributes = vec![]; + let mut gl_context_attributes = vec![]; + let mut gles_context_attributes = vec![]; + gl_context_attributes.push(khronos_egl::CONTEXT_MAJOR_VERSION); + gl_context_attributes.push(3); + gl_context_attributes.push(khronos_egl::CONTEXT_MINOR_VERSION); + gl_context_attributes.push(3); + if supports_opengl && force_gles_minor_version != wgt::Gles3MinorVersion::Automatic { + log::warn!("Ignoring specified GLES minor version as OpenGL is used"); + } + gles_context_attributes.push(khronos_egl::CONTEXT_MAJOR_VERSION); + gles_context_attributes.push(3); // Request GLES 3.0 or higher + if force_gles_minor_version != wgt::Gles3MinorVersion::Automatic { + gles_context_attributes.push(khronos_egl::CONTEXT_MINOR_VERSION); + gles_context_attributes.push(match force_gles_minor_version { + wgt::Gles3MinorVersion::Automatic => unreachable!(), + wgt::Gles3MinorVersion::Version0 => 0, + wgt::Gles3MinorVersion::Version1 => 1, + wgt::Gles3MinorVersion::Version2 => 2, + }); + } + if flags.contains(wgt::InstanceFlags::DEBUG) { + if version >= (1, 5) { + log::debug!("\tEGL context: +debug"); + context_attributes.push(khronos_egl::CONTEXT_OPENGL_DEBUG); + context_attributes.push(khronos_egl::TRUE as _); + } else if supports_khr_context { + log::debug!("\tEGL context: +debug KHR"); + khr_context_flags |= EGL_CONTEXT_OPENGL_DEBUG_BIT_KHR; + } else { + log::debug!("\tEGL context: -debug"); + } + } + + if khr_context_flags != 0 { + context_attributes.push(EGL_CONTEXT_FLAGS_KHR); + context_attributes.push(khr_context_flags); + } + + gl_context_attributes.extend(&context_attributes); + gles_context_attributes.extend(&context_attributes); + + let context = { + #[derive(Copy, Clone)] + enum Robustness { + Core, + Ext, + } + + let robustness = if version >= (1, 5) { + Some(Robustness::Core) + } else if display_extensions.contains("EGL_EXT_create_context_robustness") { + Some(Robustness::Ext) + } else { + None + }; + + let create_context = |api, base_attributes: &[khronos_egl::Int]| { + egl.bind_api(api)?; + + let mut robustness = robustness; + loop { + let robustness_attributes = match robustness { + Some(Robustness::Core) => { + vec![ + khronos_egl::CONTEXT_OPENGL_ROBUST_ACCESS, + khronos_egl::TRUE as _, + khronos_egl::NONE, + ] + } + Some(Robustness::Ext) => { + vec![ + EGL_CONTEXT_OPENGL_ROBUST_ACCESS_EXT, + khronos_egl::TRUE as _, + khronos_egl::NONE, + ] + } + None => vec![khronos_egl::NONE], + }; + + let mut context_attributes = base_attributes.to_vec(); + context_attributes.extend(&robustness_attributes); + + match egl.create_context(display, config, None, &context_attributes) { + Ok(context) => { + match robustness { + Some(Robustness::Core) => { + log::debug!("\tEGL context: +robust access"); + } + Some(Robustness::Ext) => { + log::debug!("\tEGL context: +robust access EXT"); + } + None => { + log::debug!("\tEGL context: -robust access"); + } + } + return Ok(context); + } + + // Robust access context creation can fail with different error codes + // depending on the EGL path. Retry with a lower robustness level. + Err( + khronos_egl::Error::BadAttribute + | khronos_egl::Error::BadMatch + | khronos_egl::Error::BadConfig, + ) if robustness.is_some() => { + robustness = match robustness { + Some(Robustness::Core) + if display_extensions + .contains("EGL_EXT_create_context_robustness") => + { + Some(Robustness::Ext) + } + _ => None, + }; + continue; + } + + Err(e) => return Err(e), + } + } + }; + + let result = if supports_opengl { + create_context(khronos_egl::OPENGL_API, &gl_context_attributes).or_else( + |gl_error| { + log::debug!("Failed to create desktop OpenGL context: {gl_error}, falling back to OpenGL ES"); + create_context(khronos_egl::OPENGL_ES_API, &gles_context_attributes) + }, + ) + } else { + create_context(khronos_egl::OPENGL_ES_API, &gles_context_attributes) + }; + + result.map_err(|e| { + crate::InstanceError::with_source( + String::from("unable to create OpenGL or GLES 3.x context"), + e, + ) + }) + }?; + + // Testing if context can be binded without surface + // and creating dummy pbuffer surface if not. + let pbuffer = if version >= (1, 5) + || display_extensions.contains("EGL_KHR_surfaceless_context") + || cfg!(Emscripten) + { + log::debug!("\tEGL context: +surfaceless"); + None + } else { + let attributes = [ + khronos_egl::WIDTH, + 1, + khronos_egl::HEIGHT, + 1, + khronos_egl::NONE, + ]; + egl.create_pbuffer_surface(display, config, &attributes) + .map(Some) + .map_err(|e| { + crate::InstanceError::with_source( + String::from("error in create_pbuffer_surface"), + e, + ) + })? + }; + + Ok(Self { + egl: EglContext { + instance: egl, + display, + raw: context, + pbuffer, + version, + }, + version, + supports_native_window, + config, + srgb_kind, + }) + } +} + +impl Drop for Inner { + fn drop(&mut self) { + // ERROR: Since EglContext is erroneously Clone, these handles could be copied and + // accidentally used elsewhere outside of Inner, despite us assuming ownership and + // destroying the handles here. + if let Err(e) = self + .egl + .instance + .destroy_context(self.egl.display, self.egl.raw) + { + log::warn!("Error in destroy_context: {e:?}"); + } + + if let Err(e) = terminate_display(&self.egl.instance, self.egl.display) { + log::warn!("Error in terminate: {e:?}"); + } + } +} + +#[derive(Clone, Copy, Debug, PartialEq)] +enum WindowKind { + Wayland, + X11, + AngleX11, + Unknown, +} + +#[derive(Clone, Debug)] +struct WindowSystemInterface { + kind: WindowKind, +} + +pub struct Instance { + wsi: WindowSystemInterface, + flags: wgt::InstanceFlags, + options: wgt::GlBackendOptions, + inner: Mutex, +} + +impl Instance { + pub fn raw_display(&self) -> khronos_egl::Display { + self.inner + .try_lock() + .expect("Could not lock instance. This is most-likely a deadlock.") + .egl + .display + } + + /// Returns the version of the EGL display. + pub fn egl_version(&self) -> (i32, i32) { + self.inner + .try_lock() + .expect("Could not lock instance. This is most-likely a deadlock.") + .version + } + + pub fn egl_config(&self) -> khronos_egl::Config { + self.inner + .try_lock() + .expect("Could not lock instance. This is most-likely a deadlock.") + .config + } +} + +unsafe impl Send for Instance {} +unsafe impl Sync for Instance {} + +impl crate::Instance for Instance { + type A = super::Api; + + unsafe fn init(desc: &crate::InstanceDescriptor<'_>) -> Result { + use raw_window_handle::RawDisplayHandle as Rdh; + + profiling::scope!("Init OpenGL (EGL) Backend"); + #[cfg(Emscripten)] + let egl_result: Result = + Ok(khronos_egl::Instance::new(khronos_egl::Static)); + + #[cfg(not(Emscripten))] + let egl_result = if cfg!(windows) { + unsafe { + khronos_egl::DynamicInstance::::load_required_from_filename( + "libEGL.dll", + ) + } + } else if cfg!(target_vendor = "apple") { + unsafe { + khronos_egl::DynamicInstance::::load_required_from_filename( + "libEGL.dylib", + ) + } + } else { + unsafe { khronos_egl::DynamicInstance::::load_required() } + }; + let egl = egl_result + .map(Arc::new) + .map_err(instance_err("unable to open libEGL"))?; + + let client_extensions = egl.query_string(None, khronos_egl::EXTENSIONS); + + let client_ext_str = match client_extensions { + Ok(ext) => ext.to_string_lossy().into_owned(), + Err(_) => String::new(), + }; + log::debug!( + "Client extensions: {:#?}", + client_ext_str.split_whitespace().collect::>() + ); + + #[cfg(not(Emscripten))] + let egl1_5 = egl.upcast::(); + + #[cfg(Emscripten)] + let egl1_5: Option<&Arc> = Some(&egl); + + let (display, wsi_kind) = match (desc.display.map(|d| d.as_raw()), egl1_5) { + (Some(Rdh::Wayland(wayland_display_handle)), Some(egl)) + if client_ext_str.contains("EGL_EXT_platform_wayland") => + { + log::debug!("Using Wayland platform"); + let display_attributes = [khronos_egl::ATTRIB_NONE]; + let display = unsafe { + egl.get_platform_display( + EGL_PLATFORM_WAYLAND_KHR, + wayland_display_handle.display.as_ptr(), + &display_attributes, + ) + } + .map_err(instance_err("failed to get Wayland display"))?; + (display, WindowKind::Wayland) + } + (Some(Rdh::Xlib(xlib_display_handle)), Some(egl)) + if client_ext_str.contains("EGL_EXT_platform_x11") => + { + log::debug!("Using X11 platform"); + let display_attributes = [khronos_egl::ATTRIB_NONE]; + let display = unsafe { + egl.get_platform_display( + EGL_PLATFORM_X11_KHR, + xlib_display_handle + .display + .map_or(khronos_egl::DEFAULT_DISPLAY, ptr::NonNull::as_ptr), + &display_attributes, + ) + } + .map_err(instance_err("failed to get X11 display"))?; + (display, WindowKind::X11) + } + (Some(Rdh::Xlib(xlib_display_handle)), Some(egl)) + if client_ext_str.contains("EGL_ANGLE_platform_angle") => + { + log::debug!("Using Angle platform with X11"); + let display_attributes = [ + EGL_PLATFORM_ANGLE_NATIVE_PLATFORM_TYPE_ANGLE as khronos_egl::Attrib, + EGL_PLATFORM_X11_KHR as khronos_egl::Attrib, + EGL_PLATFORM_ANGLE_DEBUG_LAYERS_ENABLED as khronos_egl::Attrib, + usize::from(desc.flags.contains(wgt::InstanceFlags::VALIDATION)), + khronos_egl::ATTRIB_NONE, + ]; + let display = unsafe { + egl.get_platform_display( + EGL_PLATFORM_ANGLE_ANGLE, + xlib_display_handle + .display + .map_or(khronos_egl::DEFAULT_DISPLAY, ptr::NonNull::as_ptr), + &display_attributes, + ) + } + .map_err(instance_err("failed to get Angle display"))?; + (display, WindowKind::AngleX11) + } + (Some(Rdh::Xcb(xcb_display_handle)), Some(egl)) + if client_ext_str.contains("EGL_EXT_platform_xcb") => + { + log::debug!("Using XCB platform"); + let display_attributes = [ + EGL_PLATFORM_XCB_SCREEN_EXT as khronos_egl::Attrib, + xcb_display_handle.screen as khronos_egl::Attrib, + khronos_egl::ATTRIB_NONE, + ]; + let display = unsafe { + egl.get_platform_display( + EGL_PLATFORM_XCB_EXT, + xcb_display_handle + .connection + .map_or(khronos_egl::DEFAULT_DISPLAY, ptr::NonNull::as_ptr), + &display_attributes, + ) + } + .map_err(instance_err("failed to get XCB display"))?; + (display, WindowKind::X11) + } + x if client_ext_str.contains("EGL_MESA_platform_surfaceless") => { + log::debug!( + "No (or unknown) windowing system ({x:?}) present. Using surfaceless platform" + ); + #[allow(clippy::unnecessary_literal_unwrap)] + // This is only a literal on Emscripten + // TODO: This extension is also supported on EGL 1.4 with EGL_EXT_platform_base: https://registry.khronos.org/EGL/extensions/MESA/EGL_MESA_platform_surfaceless.txt + let egl = egl1_5.expect("Failed to get EGL 1.5 for surfaceless"); + let display = unsafe { + egl.get_platform_display( + EGL_PLATFORM_SURFACELESS_MESA, + khronos_egl::DEFAULT_DISPLAY, + &[khronos_egl::ATTRIB_NONE], + ) + } + .map_err(instance_err("failed to get MESA surfaceless display"))?; + (display, WindowKind::Unknown) + } + x => { + log::debug!( + "No (or unknown) windowing system {x:?} and EGL_MESA_platform_surfaceless not available. Using default platform" + ); + let display = + unsafe { egl.get_display(khronos_egl::DEFAULT_DISPLAY) }.ok_or_else(|| { + crate::InstanceError::new("Failed to get default display".into()) + })?; + (display, WindowKind::Unknown) + } + }; + + if desc.flags.contains(wgt::InstanceFlags::VALIDATION) + && client_ext_str.contains("EGL_KHR_debug") + { + log::debug!("Enabling EGL debug output"); + let function: EglDebugMessageControlFun = { + let addr = egl + .get_proc_address("eglDebugMessageControlKHR") + .ok_or_else(|| { + crate::InstanceError::new( + "failed to get `eglDebugMessageControlKHR` proc address".into(), + ) + })?; + unsafe { core::mem::transmute(addr) } + }; + let attributes = [ + EGL_DEBUG_MSG_CRITICAL_KHR as khronos_egl::Attrib, + 1, + EGL_DEBUG_MSG_ERROR_KHR as khronos_egl::Attrib, + 1, + EGL_DEBUG_MSG_WARN_KHR as khronos_egl::Attrib, + 1, + EGL_DEBUG_MSG_INFO_KHR as khronos_egl::Attrib, + 1, + khronos_egl::ATTRIB_NONE, + ]; + unsafe { (function)(Some(egl_debug_proc), attributes.as_ptr()) }; + } + + let inner = Inner::create( + desc.flags, + egl, + display, + desc.backend_options.gl.gles_minor_version, + )?; + + Ok(Instance { + wsi: WindowSystemInterface { kind: wsi_kind }, + flags: desc.flags, + options: desc.backend_options.gl.clone(), + inner: Mutex::new(inner), + }) + } + + unsafe fn create_surface( + &self, + display_handle: raw_window_handle::RawDisplayHandle, + window_handle: raw_window_handle::RawWindowHandle, + ) -> Result { + use raw_window_handle::RawWindowHandle as Rwh; + + let inner = self.inner.lock(); + + match (window_handle, display_handle) { + (Rwh::Xlib(_), _) => {} + (Rwh::Xcb(_), _) => {} + (Rwh::Win32(_), _) => {} + (Rwh::AppKit(_), _) => {} + (Rwh::OhosNdk(_), _) => {} + #[cfg(target_os = "android")] + (Rwh::AndroidNdk(handle), _) => { + let format = inner + .egl + .instance + .get_config_attrib( + inner.egl.display, + inner.config, + khronos_egl::NATIVE_VISUAL_ID, + ) + .map_err(instance_err("failed to get config NATIVE_VISUAL_ID"))?; + + let ret = unsafe { + ndk_sys::ANativeWindow_setBuffersGeometry( + handle + .a_native_window + .as_ptr() + .cast::(), + 0, + 0, + format, + ) + }; + + if ret != 0 { + return Err(crate::InstanceError::new(format!( + "error {ret} returned from ANativeWindow_setBuffersGeometry", + ))); + } + } + (Rwh::Wayland(_), _) => {} + #[cfg(Emscripten)] + (Rwh::Web(_), _) => {} + other => { + return Err(crate::InstanceError::new(format!( + "unsupported window: {other:?}" + ))); + } + }; + + inner.egl.unmake_current(); + + Ok(Surface { + egl: inner.egl.clone(), + wsi: self.wsi.clone(), + config: inner.config, + presentable: inner.supports_native_window, + raw_window_handle: window_handle, + swapchain: RwLock::new(None), + srgb_kind: inner.srgb_kind, + }) + } + + unsafe fn enumerate_adapters( + &self, + _surface_hint: Option<&Surface>, + ) -> Vec> { + let inner = self.inner.lock(); + inner.egl.make_current(); + + let mut gl = unsafe { + glow::Context::from_loader_function(|name| { + inner + .egl + .instance + .get_proc_address(name) + .map_or(ptr::null(), |p| p as *const _) + }) + }; + + // In contrast to OpenGL ES, OpenGL requires explicitly enabling sRGB conversions, + // as otherwise the user has to do the sRGB conversion. + if !matches!(inner.srgb_kind, SrgbFrameBufferKind::None) { + unsafe { gl.enable(glow::FRAMEBUFFER_SRGB) }; + } + + if self.flags.contains(wgt::InstanceFlags::DEBUG) && gl.supports_debug() { + log::debug!("Max label length: {}", unsafe { + gl.get_parameter_i32(glow::MAX_LABEL_LENGTH) + }); + } + + if self.flags.contains(wgt::InstanceFlags::VALIDATION) && gl.supports_debug() { + log::debug!("Enabling GLES debug output"); + unsafe { gl.enable(glow::DEBUG_OUTPUT) }; + unsafe { gl.debug_message_callback(super::gl_debug_message_callback) }; + } + + // Wrap in ManuallyDrop to make it easier to "current" the GL context before dropping this + // GLOW context, which could also happen if a panic occurs after we uncurrent the context + // below but before AdapterContext is constructed. + let gl = ManuallyDrop::new(gl); + inner.egl.unmake_current(); + + unsafe { + super::Adapter::expose( + AdapterContext { + glow: Mutex::new(gl), + // ERROR: Copying owned reference handles here, be careful to not drop them! + egl: Some(inner.egl.clone()), + }, + self.options.clone(), + ) + } + .into_iter() + .collect() + } +} + +impl super::Adapter { + /// Creates a new external adapter using the specified loader function. + /// + /// # Safety + /// + /// - The underlying OpenGL ES context must be current. + /// - The underlying OpenGL ES context must be current when interfacing with any objects returned by + /// wgpu-hal from this adapter. + /// - The underlying OpenGL ES context must be current when dropping this adapter and when + /// dropping any objects returned from this adapter. + pub unsafe fn new_external( + fun: impl FnMut(&str) -> *const ffi::c_void, + options: wgt::GlBackendOptions, + ) -> Option> { + let context = unsafe { glow::Context::from_loader_function(fun) }; + unsafe { + Self::expose( + AdapterContext { + glow: Mutex::new(ManuallyDrop::new(context)), + egl: None, + }, + options, + ) + } + } + + pub fn adapter_context(&self) -> &AdapterContext { + &self.shared.context + } +} + +impl super::Device { + /// Returns the underlying EGL context. + pub fn context(&self) -> &AdapterContext { + &self.shared.context + } +} + +#[derive(Debug)] +pub struct Swapchain { + surface: khronos_egl::Surface, + wl_window: Option<*mut wayland_sys::egl::wl_egl_window>, + framebuffer: glow::Framebuffer, + renderbuffer: glow::Renderbuffer, + /// Extent because the window lies + extent: wgt::Extent3d, + format: wgt::TextureFormat, + format_desc: super::TextureFormatDesc, + #[allow(unused)] + sample_type: wgt::TextureSampleType, +} + +#[derive(Debug)] +pub struct Surface { + egl: EglContext, + wsi: WindowSystemInterface, + config: khronos_egl::Config, + pub(super) presentable: bool, + raw_window_handle: raw_window_handle::RawWindowHandle, + swapchain: RwLock>, + srgb_kind: SrgbFrameBufferKind, +} + +unsafe impl Send for Surface {} +unsafe impl Sync for Surface {} + +impl Surface { + pub(super) unsafe fn present( + &self, + _suf_texture: super::Texture, + context: &AdapterContext, + ) -> Result<(), crate::SurfaceError> { + let gl = unsafe { context.get_without_egl_lock() }; + let swapchain = self.swapchain.read(); + let sc = swapchain.as_ref().ok_or(crate::SurfaceError::Other( + "Surface has no swap-chain configured", + ))?; + + self.egl + .instance + .make_current( + self.egl.display, + Some(sc.surface), + Some(sc.surface), + Some(self.egl.raw), + ) + .map_err(|e| { + log::error!("make_current(surface) failed: {e}"); + crate::SurfaceError::Lost + })?; + + unsafe { gl.disable(glow::SCISSOR_TEST) }; + unsafe { gl.color_mask(true, true, true, true) }; + + unsafe { gl.bind_framebuffer(glow::DRAW_FRAMEBUFFER, None) }; + unsafe { gl.bind_framebuffer(glow::READ_FRAMEBUFFER, Some(sc.framebuffer)) }; + + if !matches!(self.srgb_kind, SrgbFrameBufferKind::None) { + // Disable sRGB conversions for `glBlitFramebuffer` as behavior does diverge between + // drivers and formats otherwise and we want to ensure no sRGB conversions happen. + unsafe { gl.disable(glow::FRAMEBUFFER_SRGB) }; + } + + // Note the Y-flipping here. GL's presentation is not flipped, + // but main rendering is. Therefore, we Y-flip the output positions + // in the shader, and also this blit. + unsafe { + gl.blit_framebuffer( + 0, + sc.extent.height as i32, + sc.extent.width as i32, + 0, + 0, + 0, + sc.extent.width as i32, + sc.extent.height as i32, + glow::COLOR_BUFFER_BIT, + glow::NEAREST, + ) + }; + + if !matches!(self.srgb_kind, SrgbFrameBufferKind::None) { + unsafe { gl.enable(glow::FRAMEBUFFER_SRGB) }; + } + + unsafe { gl.bind_framebuffer(glow::READ_FRAMEBUFFER, None) }; + + self.egl + .instance + .swap_buffers(self.egl.display, sc.surface) + .map_err(|e| { + log::error!("swap_buffers failed: {e}"); + crate::SurfaceError::Lost + // TODO: should we unset the current context here? + })?; + self.egl + .instance + .make_current(self.egl.display, None, None, None) + .map_err(|e| { + log::error!("make_current(null) failed: {e}"); + crate::SurfaceError::Lost + })?; + + Ok(()) + } + + unsafe fn unconfigure_impl( + &self, + device: &super::Device, + ) -> Option<( + khronos_egl::Surface, + Option<*mut wayland_sys::egl::wl_egl_window>, + )> { + let gl = &device.shared.context.lock(); + match self.swapchain.write().take() { + Some(sc) => { + unsafe { gl.delete_renderbuffer(sc.renderbuffer) }; + unsafe { gl.delete_framebuffer(sc.framebuffer) }; + Some((sc.surface, sc.wl_window)) + } + None => None, + } + } + + pub fn supports_srgb(&self) -> bool { + match self.srgb_kind { + SrgbFrameBufferKind::None => false, + _ => true, + } + } +} + +impl crate::Surface for Surface { + type A = super::Api; + + unsafe fn configure( + &self, + device: &super::Device, + config: &crate::SurfaceConfiguration, + ) -> Result<(), crate::SurfaceError> { + use raw_window_handle::RawWindowHandle as Rwh; + + let (surface, wl_window) = match unsafe { self.unconfigure_impl(device) } { + Some((sc, wl_window)) => { + if let Some(window) = wl_window { + wayland_sys::ffi_dispatch!( + wayland_sys::egl::wayland_egl_handle(), + wl_egl_window_resize, + window, + config.extent.width as i32, + config.extent.height as i32, + 0, + 0, + ); + } + + (sc, wl_window) + } + None => { + let mut wl_window = None; + let (mut temp_xlib_handle, mut temp_xcb_handle); + let native_window_ptr = match (self.wsi.kind, self.raw_window_handle) { + (WindowKind::Unknown | WindowKind::X11, Rwh::Xlib(handle)) => { + temp_xlib_handle = handle.window; + ptr::from_mut(&mut temp_xlib_handle).cast::() + } + (WindowKind::AngleX11, Rwh::Xlib(handle)) => handle.window as *mut ffi::c_void, + (WindowKind::Unknown | WindowKind::X11, Rwh::Xcb(handle)) => { + temp_xcb_handle = handle.window; + ptr::from_mut(&mut temp_xcb_handle).cast::() + } + (WindowKind::AngleX11, Rwh::Xcb(handle)) => { + handle.window.get() as *mut ffi::c_void + } + (WindowKind::Unknown, Rwh::AndroidNdk(handle)) => { + handle.a_native_window.as_ptr() + } + (WindowKind::Unknown, Rwh::OhosNdk(handle)) => handle.native_window.as_ptr(), + #[cfg(unix)] + (WindowKind::Wayland, Rwh::Wayland(handle)) => { + let window = wayland_sys::ffi_dispatch!( + wayland_sys::egl::wayland_egl_handle(), + wl_egl_window_create, + handle.surface.as_ptr().cast(), + config.extent.width as i32, + config.extent.height as i32, + ); + wl_window = Some(window); + window.cast() + } + #[cfg(Emscripten)] + (WindowKind::Unknown, Rwh::Web(handle)) => handle.id as *mut ffi::c_void, + (WindowKind::Unknown, Rwh::Win32(handle)) => { + handle.hwnd.get() as *mut ffi::c_void + } + (WindowKind::Unknown, Rwh::AppKit(handle)) => { + #[cfg(not(target_os = "macos"))] + let window_ptr = handle.ns_view.as_ptr(); + #[cfg(target_os = "macos")] + let window_ptr = { + use objc2::msg_send; + use objc2::runtime::AnyObject; + // ns_view always have a layer and don't need to verify that it exists. + let layer: *mut AnyObject = + msg_send![handle.ns_view.as_ptr().cast::(), layer]; + layer.cast::() + }; + window_ptr + } + _ => { + log::warn!( + "Initialized platform {:?} doesn't work with window {:?}", + self.wsi.kind, + self.raw_window_handle + ); + return Err(crate::SurfaceError::Other("incompatible window kind")); + } + }; + + let mut attributes = vec![ + khronos_egl::RENDER_BUFFER, + // We don't want any of the buffering done by the driver, because we + // manage a swapchain on our side. + // Some drivers just fail on surface creation seeing `EGL_SINGLE_BUFFER`. + if cfg!(any( + target_os = "android", + target_os = "macos", + target_env = "ohos" + )) || cfg!(windows) + || self.wsi.kind == WindowKind::AngleX11 + { + khronos_egl::BACK_BUFFER + } else { + khronos_egl::SINGLE_BUFFER + }, + ]; + if config.format.is_srgb() { + match self.srgb_kind { + SrgbFrameBufferKind::None => {} + SrgbFrameBufferKind::Core => { + attributes.push(khronos_egl::GL_COLORSPACE); + attributes.push(khronos_egl::GL_COLORSPACE_SRGB); + } + SrgbFrameBufferKind::Khr => { + attributes.push(EGL_GL_COLORSPACE_KHR as i32); + attributes.push(EGL_GL_COLORSPACE_SRGB_KHR as i32); + } + } + } + attributes.push(khronos_egl::ATTRIB_NONE as i32); + + #[cfg(not(Emscripten))] + let egl1_5 = self.egl.instance.upcast::(); + + #[cfg(Emscripten)] + let egl1_5: Option<&Arc> = Some(&self.egl.instance); + + // Careful, we can still be in 1.4 version even if `upcast` succeeds + let raw_result = match egl1_5 { + Some(egl) if self.wsi.kind != WindowKind::Unknown => { + let attributes_usize = attributes + .into_iter() + .map(|v| v as usize) + .collect::>(); + unsafe { + egl.create_platform_window_surface( + self.egl.display, + self.config, + native_window_ptr, + &attributes_usize, + ) + } + } + _ => unsafe { + self.egl.instance.create_window_surface( + self.egl.display, + self.config, + native_window_ptr, + Some(&attributes), + ) + }, + }; + + match raw_result { + Ok(raw) => (raw, wl_window), + Err(e) => { + log::warn!("Error in create_window_surface: {e:?}"); + return Err(crate::SurfaceError::Lost); + } + } + } + }; + + let format_desc = device.shared.describe_texture_format(config.format); + let gl = &device.shared.context.lock(); + let renderbuffer = unsafe { gl.create_renderbuffer() }.map_err(|error| { + log::error!("Internal swapchain renderbuffer creation failed: {error}"); + crate::DeviceError::OutOfMemory + })?; + unsafe { gl.bind_renderbuffer(glow::RENDERBUFFER, Some(renderbuffer)) }; + unsafe { + gl.renderbuffer_storage( + glow::RENDERBUFFER, + format_desc.internal, + config.extent.width as _, + config.extent.height as _, + ) + }; + let framebuffer = unsafe { gl.create_framebuffer() }.map_err(|error| { + log::error!("Internal swapchain framebuffer creation failed: {error}"); + crate::DeviceError::OutOfMemory + })?; + unsafe { gl.bind_framebuffer(glow::READ_FRAMEBUFFER, Some(framebuffer)) }; + unsafe { + gl.framebuffer_renderbuffer( + glow::READ_FRAMEBUFFER, + glow::COLOR_ATTACHMENT0, + glow::RENDERBUFFER, + Some(renderbuffer), + ) + }; + unsafe { gl.bind_renderbuffer(glow::RENDERBUFFER, None) }; + unsafe { gl.bind_framebuffer(glow::READ_FRAMEBUFFER, None) }; + + let mut swapchain = self.swapchain.write(); + *swapchain = Some(Swapchain { + surface, + wl_window, + renderbuffer, + framebuffer, + extent: config.extent, + format: config.format, + format_desc, + sample_type: wgt::TextureSampleType::Float { filterable: false }, + }); + + Ok(()) + } + + unsafe fn unconfigure(&self, device: &super::Device) { + if let Some((surface, wl_window)) = unsafe { self.unconfigure_impl(device) } { + self.egl + .instance + .destroy_surface(self.egl.display, surface) + .unwrap(); + if let Some(window) = wl_window { + wayland_sys::ffi_dispatch!( + wayland_sys::egl::wayland_egl_handle(), + wl_egl_window_destroy, + window, + ); + } + } + } + + unsafe fn acquire_texture( + &self, + _timeout_ms: Option, //TODO + _fence: &super::Fence, + ) -> Result, crate::SurfaceError> { + let swapchain = self.swapchain.read(); + let sc = swapchain.as_ref().ok_or(crate::SurfaceError::Other( + "Surface has no swap-chain configured", + ))?; + let texture = super::Texture { + inner: super::TextureInner::Renderbuffer { + raw: sc.renderbuffer, + }, + drop_guard: None, + array_layer_count: 1, + mip_level_count: 1, + format: sc.format, + format_desc: sc.format_desc.clone(), + copy_size: crate::CopyExtent { + width: sc.extent.width, + height: sc.extent.height, + depth: 1, + }, + }; + Ok(crate::AcquiredSurfaceTexture { + texture, + suboptimal: false, + }) + } + unsafe fn discard_texture(&self, _texture: super::Texture) {} +} diff --git a/third_party/wgpu-hal-29.0.4/src/gles/emscripten.rs b/third_party/wgpu-hal-29.0.4/src/gles/emscripten.rs new file mode 100644 index 0000000..88cc0be --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/emscripten.rs @@ -0,0 +1,26 @@ +extern "C" { + /// returns 1 if success. 0 if failure. extension name must be null terminated + fn emscripten_webgl_enable_extension( + context: core::ffi::c_int, + extension: *const core::ffi::c_char, + ) -> core::ffi::c_int; + fn emscripten_webgl_get_current_context() -> core::ffi::c_int; +} +/// Webgl requires extensions to be enabled before using them. +/// This function can be used to enable webgl extension on emscripten target. +/// +/// returns true on success +/// +/// # Safety +/// +/// - opengl context MUST BE current +/// - extension_name_null_terminated argument must be a valid string with null terminator. +/// - extension must be present. check `glow_context.supported_extensions()` +pub unsafe fn enable_extension(extension_name_null_terminated: &str) -> bool { + unsafe { + emscripten_webgl_enable_extension( + emscripten_webgl_get_current_context(), + extension_name_null_terminated.as_ptr().cast(), + ) == 1 + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/gles/fence.rs b/third_party/wgpu-hal-29.0.4/src/gles/fence.rs new file mode 100644 index 0000000..d5cd0ec --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/fence.rs @@ -0,0 +1,168 @@ +use alloc::vec::Vec; +use core::sync::atomic::Ordering; + +use glow::HasContext; + +use crate::AtomicFenceValue; + +#[derive(Debug, Copy, Clone)] +struct GLFence { + sync: glow::Fence, + value: crate::FenceValue, +} + +#[derive(Debug)] +pub struct Fence { + last_completed: AtomicFenceValue, + pending: Vec, + fence_behavior: wgt::GlFenceBehavior, +} + +impl crate::DynFence for Fence {} + +#[cfg(send_sync)] +unsafe impl Send for Fence {} +#[cfg(send_sync)] +unsafe impl Sync for Fence {} + +impl Fence { + pub fn new(options: &wgt::GlBackendOptions) -> Self { + Self { + last_completed: AtomicFenceValue::new(0), + pending: Vec::new(), + fence_behavior: options.fence_behavior, + } + } + + pub fn signal( + &mut self, + gl: &glow::Context, + value: crate::FenceValue, + ) -> Result<(), crate::DeviceError> { + if self.fence_behavior.is_auto_finish() { + *self.last_completed.get_mut() = value; + return Ok(()); + } + + let sync = unsafe { gl.fence_sync(glow::SYNC_GPU_COMMANDS_COMPLETE, 0) } + .map_err(|_| crate::DeviceError::OutOfMemory)?; + self.pending.push(GLFence { sync, value }); + + Ok(()) + } + + pub fn satisfied(&self, value: crate::FenceValue) -> bool { + self.last_completed.load(Ordering::Acquire) >= value + } + + pub fn get_latest(&self, gl: &glow::Context) -> crate::FenceValue { + let mut max_value = self.last_completed.load(Ordering::Acquire); + + if self.fence_behavior.is_auto_finish() { + return max_value; + } + + for gl_fence in self.pending.iter() { + if gl_fence.value <= max_value { + // We already know this was good, no need to check again + continue; + } + let status = unsafe { gl.get_sync_status(gl_fence.sync) }; + if status == glow::SIGNALED { + max_value = gl_fence.value; + } else { + // Anything after the first unsignalled is guaranteed to also be unsignalled + break; + } + } + + // Track the latest value, to save ourselves some querying later + self.last_completed.fetch_max(max_value, Ordering::AcqRel); + + max_value + } + + pub fn maintain(&mut self, gl: &glow::Context) { + if self.fence_behavior.is_auto_finish() { + return; + } + + let latest = self.get_latest(gl); + for &gl_fence in self.pending.iter() { + if gl_fence.value <= latest { + unsafe { + gl.delete_sync(gl_fence.sync); + } + } + } + self.pending.retain(|&gl_fence| gl_fence.value > latest); + } + + pub fn wait( + &self, + gl: &glow::Context, + wait_value: crate::FenceValue, + timeout_ns: u32, + ) -> Result { + let last_completed = self.last_completed.load(Ordering::Acquire); + + if self.fence_behavior.is_auto_finish() { + return Ok(last_completed >= wait_value); + } + + // We already know this fence has been signalled to that value. Return signalled. + if last_completed >= wait_value { + return Ok(true); + } + + // Find a matching fence + let gl_fence = self + .pending + .iter() + // Greater or equal as an abundance of caution, but there should be one fence per value + .find(|gl_fence| gl_fence.value >= wait_value); + + let Some(gl_fence) = gl_fence else { + log::warn!("Tried to wait for {wait_value} but that value has not been signalled yet"); + return Ok(false); + }; + + // We should have found a fence with the exact value. + debug_assert_eq!(gl_fence.value, wait_value); + + let status = unsafe { + gl.client_wait_sync( + gl_fence.sync, + glow::SYNC_FLUSH_COMMANDS_BIT, + timeout_ns.min(i32::MAX as u32) as i32, + ) + }; + + let signalled = match status { + glow::ALREADY_SIGNALED | glow::CONDITION_SATISFIED => true, + glow::TIMEOUT_EXPIRED | glow::WAIT_FAILED => false, + _ => { + log::warn!("Unexpected result from client_wait_sync: {status}"); + false + } + }; + + if signalled { + self.last_completed.fetch_max(wait_value, Ordering::AcqRel); + } + + Ok(signalled) + } + + pub fn destroy(self, gl: &glow::Context) { + if self.fence_behavior.is_auto_finish() { + return; + } + + for gl_fence in self.pending { + unsafe { + gl.delete_sync(gl_fence.sync); + } + } + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/gles/mod.rs b/third_party/wgpu-hal-29.0.4/src/gles/mod.rs new file mode 100644 index 0000000..6f8c865 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/mod.rs @@ -0,0 +1,1140 @@ +/*! +# OpenGL ES3 API (aka GLES3). + +Designed to work on Linux and Android, with context provided by EGL. + +## Texture views + +GLES3 doesn't really have separate texture view objects. We have to remember the +original texture and the sub-range into it. Problem is, however, that there is +no way to expose a subset of array layers or mip levels of a sampled texture. + +## Binding model + +Binding model is very different from WebGPU, especially with regards to samplers. +GLES3 has sampler objects, but they aren't separately bindable to the shaders. +Each sampled texture is exposed to the shader as a combined texture-sampler binding. + +When building the pipeline layout, we linearize binding entries based on the groups +(uniform/storage buffers, uniform/storage textures), and record the mapping into +`BindGroupLayoutInfo`. +When a pipeline gets created, and we track all the texture-sampler associations +from the static use in the shader. +We only support at most one sampler used with each texture so far. The linear index +of this sampler is stored per texture slot in `SamplerBindMap` array. + +The texture-sampler pairs get potentially invalidated in 2 places: + - when a new pipeline is set, we update the linear indices of associated samplers + - when a new bind group is set, we update both the textures and the samplers + +We expect that the changes to sampler states between any 2 pipelines of the same layout +will be minimal, if any. + +## Vertex data + +Generally, vertex buffers are marked as dirty and lazily bound on draw. + +GLES3 doesn't support `first_instance` semantics. However, it's easy to support, +since we are forced to do late binding anyway. We just adjust the offsets +into the vertex data. + +### Old path + +In GLES-3.0 and WebGL2, vertex buffer layout is provided +together with the actual buffer binding. +We invalidate the attributes on the vertex buffer change, and re-bind them. + +### New path + +In GLES-3.1 and higher, the vertex buffer layout can be declared separately +from the vertex data itself. This mostly matches WebGPU, however there is a catch: +`stride` needs to be specified with the data, not as a part of the layout. + +To address this, we invalidate the vertex buffers based on: + - whether or not `first_instance` is used + - stride has changed + +## Handling of `base_vertex`, `first_instance`, and `first_vertex` + +Between indirect, the lack of `first_instance` semantics, and the availability of `gl_BaseInstance` +in shaders, getting buffers and builtins to work correctly is a bit tricky. + +We never emulate `base_vertex` and gl_VertexID behaves as `@builtin(vertex_index)` does, so we +never need to do anything about that. + +### GL 4.2+ with ARB_shader_draw_parameters + +- `@builtin(instance_index)` translates to `gl_InstanceID + gl_BaseInstance` +- We bind instance buffers without any offset emulation. +- We advertise support for the `INDIRECT_FIRST_INSTANCE` feature. + +While we can theoretically have a card with 4.2+ support but without ARB_shader_draw_parameters, +we don't bother with that combination. + +### GLES & GL 4.1 + +- `@builtin(instance_index)` translates to `gl_InstanceID + naga_vs_first_instance` +- We bind instance buffers with offset emulation. +- We _do not_ advertise support for `INDIRECT_FIRST_INSTANCE` and cpu-side pretend the `first_instance` is 0 on indirect calls. + +*/ + +///cbindgen:ignore +#[cfg(not(any(windows, webgl)))] +mod egl; +#[cfg(Emscripten)] +mod emscripten; +#[cfg(webgl)] +mod web; +#[cfg(windows)] +mod wgl; + +mod adapter; +mod command; +mod conv; +mod device; +mod fence; +mod queue; + +pub use fence::Fence; + +#[cfg(not(any(windows, webgl)))] +pub use self::egl::{AdapterContext, AdapterContextLock}; +#[cfg(not(any(windows, webgl)))] +pub use self::egl::{Instance, Surface}; + +#[cfg(webgl)] +pub use self::web::AdapterContext; +#[cfg(webgl)] +pub use self::web::{Instance, Surface}; + +#[cfg(windows)] +use self::wgl::AdapterContext; +#[cfg(windows)] +pub use self::wgl::{Instance, Surface}; + +use alloc::{boxed::Box, string::String, string::ToString as _, sync::Arc, vec::Vec}; +use core::{ + fmt, + ops::Range, + sync::atomic::{AtomicU32, AtomicU8}, +}; +use parking_lot::Mutex; + +use arrayvec::ArrayVec; +use glow::HasContext; +use naga::FastHashMap; + +use crate::{CopyExtent, TextureDescriptor}; + +#[derive(Clone, Debug)] +pub struct Api; + +//Note: we can support more samplers if not every one of them is used at a time, +// but it probably doesn't worth it. +const MAX_TEXTURE_SLOTS: usize = 16; +const MAX_SAMPLERS: usize = 16; +const MAX_VERTEX_ATTRIBUTES: usize = 16; +const ZERO_BUFFER_SIZE: usize = 256 << 10; +const MAX_IMMEDIATES: usize = 64; +// We have to account for each immediate data may need to be set for every shader. +const MAX_IMMEDIATES_COMMANDS: usize = MAX_IMMEDIATES * crate::MAX_CONCURRENT_SHADER_STAGES; + +impl crate::Api for Api { + const VARIANT: wgt::Backend = wgt::Backend::Gl; + + type Instance = Instance; + type Surface = Surface; + type Adapter = Adapter; + type Device = Device; + + type Queue = Queue; + type CommandEncoder = CommandEncoder; + type CommandBuffer = CommandBuffer; + + type Buffer = Buffer; + type Texture = Texture; + type SurfaceTexture = Texture; + type TextureView = TextureView; + type Sampler = Sampler; + type QuerySet = QuerySet; + type Fence = Fence; + type AccelerationStructure = AccelerationStructure; + type PipelineCache = PipelineCache; + + type BindGroupLayout = BindGroupLayout; + type BindGroup = BindGroup; + type PipelineLayout = PipelineLayout; + type ShaderModule = ShaderModule; + type RenderPipeline = RenderPipeline; + type ComputePipeline = ComputePipeline; +} + +crate::impl_dyn_resource!( + Adapter, + AccelerationStructure, + BindGroup, + BindGroupLayout, + Buffer, + CommandBuffer, + CommandEncoder, + ComputePipeline, + Device, + Fence, + Instance, + PipelineCache, + PipelineLayout, + QuerySet, + Queue, + RenderPipeline, + Sampler, + ShaderModule, + Surface, + Texture, + TextureView +); + +bitflags::bitflags! { + /// Flags that affect internal code paths but do not + /// change the exposed feature set. + #[derive(Debug, Copy, Clone, PartialEq, Eq, Hash)] + struct PrivateCapabilities: u32 { + /// Indicates support for `glBufferStorage` allocation. + const BUFFER_ALLOCATION = 1 << 0; + /// Support explicit layouts in shader. + const SHADER_BINDING_LAYOUT = 1 << 1; + /// Support extended shadow sampling instructions. + const SHADER_TEXTURE_SHADOW_LOD = 1 << 2; + /// Support memory barriers. + const MEMORY_BARRIERS = 1 << 3; + /// Vertex buffer layouts separate from the data. + const VERTEX_BUFFER_LAYOUT = 1 << 4; + /// Indicates that buffers used as `GL_ELEMENT_ARRAY_BUFFER` may be created / initialized / used + /// as other targets, if not present they must not be mixed with other targets. + const INDEX_BUFFER_ROLE_CHANGE = 1 << 5; + /// Supports `glGetBufferSubData` + const GET_BUFFER_SUB_DATA = 1 << 7; + /// Supports `f16` color buffers + const COLOR_BUFFER_HALF_FLOAT = 1 << 8; + /// Supports `f11/f10` and `f32` color buffers + const COLOR_BUFFER_FLOAT = 1 << 9; + /// Supports query buffer objects. + const QUERY_BUFFERS = 1 << 11; + /// Supports 64 bit queries via `glGetQueryObjectui64v` + const QUERY_64BIT = 1 << 12; + /// Supports `glTexStorage2D`, etc. + const TEXTURE_STORAGE = 1 << 13; + /// Supports `push_debug_group`, `pop_debug_group` and `debug_message_insert`. + const DEBUG_FNS = 1 << 14; + /// Supports framebuffer invalidation. + const INVALIDATE_FRAMEBUFFER = 1 << 15; + /// Indicates support for `glDrawElementsInstancedBaseVertexBaseInstance` and `ARB_shader_draw_parameters` + /// + /// When this is true, instance offset emulation via vertex buffer rebinding and a shader uniform will be disabled. + const FULLY_FEATURED_INSTANCING = 1 << 16; + /// Supports direct multisampled rendering to a texture without needing a resolve texture. + const MULTISAMPLED_RENDER_TO_TEXTURE = 1 << 17; + } +} + +bitflags::bitflags! { + /// Flags that indicate necessary workarounds for specific devices or driver bugs + #[derive(Debug, Copy, Clone, PartialEq, Eq, Hash)] + struct Workarounds: u32 { + // Needs workaround for Intel Mesa bug: + // https://gitlab.freedesktop.org/mesa/mesa/-/issues/2565. + // + // This comment + // (https://gitlab.freedesktop.org/mesa/mesa/-/merge_requests/4972/diffs?diff_id=75888#22f5d1004713c9bbf857988c7efb81631ab88f99_323_327) + // seems to indicate all skylake models are effected. + const MESA_I915_SRGB_SHADER_CLEAR = 1 << 0; + /// Buffer map must emulated because it is not supported natively + const EMULATE_BUFFER_MAP = 1 << 1; + } +} + +type BindTarget = u32; + +#[derive(Debug, Default, Clone, Copy)] +enum VertexAttribKind { + #[default] + Float, // glVertexAttribPointer + Integer, // glVertexAttribIPointer + //Double, // glVertexAttribLPointer +} + +#[derive(Clone, Debug)] +pub struct TextureFormatDesc { + pub internal: u32, + pub external: u32, + pub data_type: u32, +} + +struct AdapterShared { + context: AdapterContext, + private_caps: PrivateCapabilities, + features: wgt::Features, + limits: wgt::Limits, + workarounds: Workarounds, + options: wgt::GlBackendOptions, + shading_language_version: naga::back::glsl::Version, + next_shader_id: AtomicU32, + program_cache: Mutex, + es: bool, + + /// Result of `gl.get_parameter_i32(glow::MAX_SAMPLES)`. + /// Cached here so it doesn't need to be queried every time texture format capabilities are requested. + /// (this has been shown to be a significant enough overhead) + max_msaa_samples: i32, +} + +pub struct Adapter { + shared: Arc, +} + +pub struct Device { + shared: Arc, + main_vao: glow::VertexArray, + #[cfg(all(native, feature = "renderdoc"))] + render_doc: crate::auxil::renderdoc::RenderDoc, + counters: Arc, +} + +impl Drop for Device { + fn drop(&mut self) { + let gl = &self.shared.context.lock(); + unsafe { gl.delete_vertex_array(self.main_vao) }; + } +} + +pub struct ShaderClearProgram { + pub program: glow::Program, + pub color_uniform_location: glow::UniformLocation, +} + +pub struct Queue { + shared: Arc, + features: wgt::Features, + draw_fbo: glow::Framebuffer, + copy_fbo: glow::Framebuffer, + /// Shader program used to clear the screen for [`Workarounds::MESA_I915_SRGB_SHADER_CLEAR`] + /// devices. + shader_clear_program: Option, + /// Keep a reasonably large buffer filled with zeroes, so that we can implement `ClearBuffer` of + /// zeroes by copying from it. + zero_buffer: glow::Buffer, + temp_query_results: Mutex>, + draw_buffer_count: AtomicU8, + current_index_buffer: Mutex>, +} + +impl Drop for Queue { + fn drop(&mut self) { + let gl = &self.shared.context.lock(); + unsafe { gl.delete_framebuffer(self.draw_fbo) }; + unsafe { gl.delete_framebuffer(self.copy_fbo) }; + unsafe { gl.delete_buffer(self.zero_buffer) }; + } +} + +#[derive(Clone, Debug)] +pub struct Buffer { + raw: Option, + target: BindTarget, + size: wgt::BufferAddress, + /// Flags to use within calls to [`Device::map_buffer`](crate::Device::map_buffer). + map_flags: u32, + data: Option>>>, + offset_of_current_mapping: Arc>, +} + +#[cfg(send_sync)] +unsafe impl Sync for Buffer {} +#[cfg(send_sync)] +unsafe impl Send for Buffer {} + +impl crate::DynBuffer for Buffer {} + +#[derive(Clone, Debug)] +pub enum TextureInner { + Renderbuffer { + raw: glow::Renderbuffer, + }, + DefaultRenderbuffer, + Texture { + raw: glow::Texture, + target: BindTarget, + }, + #[cfg(webgl)] + /// Render to a `WebGLFramebuffer` + /// + /// This is a web feature + ExternalFramebuffer { + inner: web_sys::WebGlFramebuffer, + }, + #[cfg(native)] + /// Render to a `glow::NativeFramebuffer` + /// Useful when the framebuffer to draw to + /// has a non-zero framebuffer ID + /// + /// This is a native feature + ExternalNativeFramebuffer { + inner: glow::NativeFramebuffer, + }, +} + +#[cfg(send_sync)] +unsafe impl Sync for TextureInner {} +#[cfg(send_sync)] +unsafe impl Send for TextureInner {} + +impl TextureInner { + fn as_native(&self) -> (glow::Texture, BindTarget) { + match *self { + Self::Renderbuffer { .. } | Self::DefaultRenderbuffer => { + panic!("Unexpected renderbuffer"); + } + Self::Texture { raw, target } => (raw, target), + #[cfg(webgl)] + Self::ExternalFramebuffer { .. } => panic!("Unexpected external framebuffer"), + #[cfg(native)] + Self::ExternalNativeFramebuffer { .. } => panic!("unexpected external framebuffer"), + } + } +} + +#[derive(Debug)] +pub struct Texture { + pub inner: TextureInner, + pub mip_level_count: u32, + pub array_layer_count: u32, + pub format: wgt::TextureFormat, + pub format_desc: TextureFormatDesc, + pub copy_size: CopyExtent, + + // The `drop_guard` field must be the last field of this struct so it is dropped last. + // Do not add new fields after it. + pub drop_guard: Option, +} + +impl crate::DynTexture for Texture {} +impl crate::DynSurfaceTexture for Texture {} + +impl core::borrow::Borrow for Texture { + fn borrow(&self) -> &dyn crate::DynTexture { + self + } +} + +impl Texture { + pub fn default_framebuffer(format: wgt::TextureFormat) -> Self { + Self { + inner: TextureInner::DefaultRenderbuffer, + drop_guard: None, + mip_level_count: 1, + array_layer_count: 1, + format, + format_desc: TextureFormatDesc { + internal: 0, + external: 0, + data_type: 0, + }, + copy_size: CopyExtent { + width: 0, + height: 0, + depth: 0, + }, + } + } + + /// Returns the `target`, whether the image is 3d and whether the image is a cubemap. + fn get_info_from_desc(desc: &TextureDescriptor) -> u32 { + match desc.dimension { + // WebGL (1 and 2) as well as some GLES versions do not have 1D textures, so we are + // doing `TEXTURE_2D` instead + wgt::TextureDimension::D1 => glow::TEXTURE_2D, + wgt::TextureDimension::D2 => { + // HACK: detect a cube map; forces cube compatible textures to be cube textures + match (desc.is_cube_compatible(), desc.size.depth_or_array_layers) { + (false, 1) => glow::TEXTURE_2D, + (false, _) => glow::TEXTURE_2D_ARRAY, + (true, 6) => glow::TEXTURE_CUBE_MAP, + (true, _) => glow::TEXTURE_CUBE_MAP_ARRAY, + } + } + wgt::TextureDimension::D3 => glow::TEXTURE_3D, + } + } + + /// More information can be found in issues #1614 and #1574 + fn log_failing_target_heuristics(view_dimension: wgt::TextureViewDimension, target: u32) { + let expected_target = match view_dimension { + wgt::TextureViewDimension::D1 => glow::TEXTURE_2D, + wgt::TextureViewDimension::D2 => glow::TEXTURE_2D, + wgt::TextureViewDimension::D2Array => glow::TEXTURE_2D_ARRAY, + wgt::TextureViewDimension::Cube => glow::TEXTURE_CUBE_MAP, + wgt::TextureViewDimension::CubeArray => glow::TEXTURE_CUBE_MAP_ARRAY, + wgt::TextureViewDimension::D3 => glow::TEXTURE_3D, + }; + + if expected_target == target { + return; + } + + let buffer; + let got = match target { + glow::TEXTURE_2D => "D2", + glow::TEXTURE_2D_ARRAY => "D2Array", + glow::TEXTURE_CUBE_MAP => "Cube", + glow::TEXTURE_CUBE_MAP_ARRAY => "CubeArray", + glow::TEXTURE_3D => "D3", + target => { + buffer = target.to_string(); + &buffer + } + }; + + log::error!( + concat!( + "wgpu-hal heuristics assumed that ", + "the view dimension will be equal to `{}` rather than `{:?}`.\n", + "`D2` textures with ", + "`depth_or_array_layers == 1` ", + "are assumed to have view dimension `D2`\n", + "`D2` textures with ", + "`depth_or_array_layers > 1` ", + "are assumed to have view dimension `D2Array`\n", + "`D2` textures with ", + "`depth_or_array_layers == 6` ", + "are assumed to have view dimension `Cube`\n", + "`D2` textures with ", + "`depth_or_array_layers > 6 && depth_or_array_layers % 6 == 0` ", + "are assumed to have view dimension `CubeArray`\n", + ), + got, + view_dimension, + ); + } +} + +#[derive(Clone, Debug)] +pub struct TextureView { + inner: TextureInner, + aspects: crate::FormatAspects, + mip_levels: Range, + array_layers: Range, + format: wgt::TextureFormat, +} + +impl crate::DynTextureView for TextureView {} + +#[derive(Debug)] +pub struct Sampler { + raw: glow::Sampler, +} + +impl crate::DynSampler for Sampler {} + +#[derive(Debug)] +pub struct BindGroupLayout { + entries: Arc<[wgt::BindGroupLayoutEntry]>, +} + +impl crate::DynBindGroupLayout for BindGroupLayout {} + +#[derive(Debug)] +struct BindGroupLayoutInfo { + entries: Arc<[wgt::BindGroupLayoutEntry]>, + /// Mapping of resources, indexed by `binding`, into the whole layout space. + /// For texture resources, the value is the texture slot index. + /// For sampler resources, the value is the index of the sampler in the whole layout. + /// For buffers, the value is the uniform or storage slot index. + /// For unused bindings, the value is `!0` + binding_to_slot: Box<[u8]>, +} + +#[derive(Debug)] +pub struct PipelineLayout { + group_infos: Box<[Option]>, + naga_options: naga::back::glsl::Options, +} + +impl crate::DynPipelineLayout for PipelineLayout {} + +impl PipelineLayout { + /// # Panics + /// If the pipeline layout does not contain a bind group layout used by + /// the resource binding. + fn get_slot(&self, br: &naga::ResourceBinding) -> u8 { + let group_info = self.group_infos[br.group as usize].as_ref().unwrap(); + group_info.binding_to_slot[br.binding as usize] + } +} + +#[derive(Debug)] +enum BindingRegister { + UniformBuffers, + StorageBuffers, + Textures, + Images, +} + +#[derive(Debug)] +enum RawBinding { + Buffer { + raw: glow::Buffer, + offset: i32, + size: i32, + }, + Texture { + raw: glow::Texture, + target: BindTarget, + aspects: crate::FormatAspects, + mip_levels: Range, + //TODO: array layers + }, + Image(ImageBinding), + Sampler(glow::Sampler), +} + +#[derive(Debug)] +pub struct BindGroup { + contents: Box<[RawBinding]>, +} + +impl crate::DynBindGroup for BindGroup {} + +type ShaderId = u32; + +#[derive(Debug)] +pub struct ShaderModule { + source: crate::NagaShader, + label: Option, + id: ShaderId, +} + +impl crate::DynShaderModule for ShaderModule {} + +#[derive(Clone, Debug, Default)] +struct VertexFormatDesc { + element_count: i32, + element_format: u32, + attrib_kind: VertexAttribKind, +} + +#[derive(Clone, Debug, Default)] +struct AttributeDesc { + location: u32, + offset: u32, + buffer_index: u32, + format_desc: VertexFormatDesc, +} + +#[derive(Clone, Debug)] +struct BufferBinding { + raw: glow::Buffer, + offset: wgt::BufferAddress, +} + +#[derive(Clone, Debug)] +struct ImageBinding { + raw: glow::Texture, + mip_level: u32, + array_layer: Option, + access: u32, + format: u32, +} + +#[derive(Clone, Debug, Default, PartialEq)] +struct VertexBufferDesc { + step: wgt::VertexStepMode, + stride: u32, +} + +#[derive(Clone, Debug)] +struct ImmediateDesc { + location: glow::UniformLocation, + ty: naga::TypeInner, + offset: u32, + size_bytes: u32, +} + +#[cfg(send_sync)] +unsafe impl Sync for ImmediateDesc {} +#[cfg(send_sync)] +unsafe impl Send for ImmediateDesc {} + +/// For each texture in the pipeline layout, store the index of the only +/// sampler (in this layout) that the texture is used with. +type SamplerBindMap = [Option; MAX_TEXTURE_SLOTS]; + +#[derive(Debug)] +struct PipelineInner { + program: glow::Program, + sampler_map: SamplerBindMap, + first_instance_location: Option, + immediates_descs: ArrayVec, + clip_distance_count: u32, +} + +#[derive(Clone, Debug)] +struct DepthState { + function: u32, + mask: bool, +} + +#[derive(Clone, Debug, PartialEq)] +struct BlendComponent { + src: u32, + dst: u32, + equation: u32, +} + +#[derive(Clone, Debug, PartialEq)] +struct BlendDesc { + alpha: BlendComponent, + color: BlendComponent, +} + +#[derive(Clone, Debug, Default, PartialEq)] +struct ColorTargetDesc { + mask: wgt::ColorWrites, + blend: Option, +} + +#[derive(PartialEq, Eq, Hash)] +struct ProgramStage { + naga_stage: naga::ShaderStage, + shader_id: ShaderId, + entry_point: String, + zero_initialize_workgroup_memory: bool, + constant_hash: Vec, +} + +#[derive(PartialEq, Eq, Hash)] +struct ProgramCacheKey { + stages: ArrayVec, + group_to_binding_to_slot: Box<[Option>]>, +} + +type ProgramCache = FastHashMap, crate::PipelineError>>; + +#[derive(Debug)] +pub struct RenderPipeline { + inner: Arc, + primitive: wgt::PrimitiveState, + vertex_buffers: Box<[VertexBufferDesc]>, + vertex_attributes: Box<[AttributeDesc]>, + color_targets: Box<[ColorTargetDesc]>, + depth: Option, + depth_bias: wgt::DepthBiasState, + stencil: Option, + alpha_to_coverage_enabled: bool, +} + +impl crate::DynRenderPipeline for RenderPipeline {} + +#[cfg(send_sync)] +unsafe impl Sync for RenderPipeline {} +#[cfg(send_sync)] +unsafe impl Send for RenderPipeline {} + +#[derive(Debug)] +pub struct ComputePipeline { + inner: Arc, +} + +impl crate::DynComputePipeline for ComputePipeline {} + +#[cfg(send_sync)] +unsafe impl Sync for ComputePipeline {} +#[cfg(send_sync)] +unsafe impl Send for ComputePipeline {} + +#[derive(Debug)] +pub struct QuerySet { + queries: Box<[glow::Query]>, + target: BindTarget, +} + +impl crate::DynQuerySet for QuerySet {} + +#[derive(Debug)] +pub struct AccelerationStructure; + +impl crate::DynAccelerationStructure for AccelerationStructure {} + +#[derive(Debug)] +pub struct PipelineCache; + +impl crate::DynPipelineCache for PipelineCache {} + +#[derive(Clone, Debug, PartialEq)] +struct StencilOps { + pass: u32, + fail: u32, + depth_fail: u32, +} + +impl Default for StencilOps { + fn default() -> Self { + Self { + pass: glow::KEEP, + fail: glow::KEEP, + depth_fail: glow::KEEP, + } + } +} + +#[derive(Clone, Debug, PartialEq)] +struct StencilSide { + function: u32, + mask_read: u32, + mask_write: u32, + reference: u32, + ops: StencilOps, +} + +impl Default for StencilSide { + fn default() -> Self { + Self { + function: glow::ALWAYS, + mask_read: 0xFF, + mask_write: 0xFF, + reference: 0, + ops: StencilOps::default(), + } + } +} + +#[derive(Debug, Clone, Default)] +struct StencilState { + front: StencilSide, + back: StencilSide, +} + +#[derive(Clone, Debug, Default, PartialEq)] +struct PrimitiveState { + front_face: u32, + cull_face: u32, + unclipped_depth: bool, + polygon_mode: u32, +} + +type InvalidatedAttachments = ArrayVec; + +#[derive(Debug)] +enum Command { + Draw { + topology: u32, + first_vertex: u32, + vertex_count: u32, + first_instance: u32, + instance_count: u32, + first_instance_location: Option, + }, + DrawIndexed { + topology: u32, + index_type: u32, + index_count: u32, + index_offset: wgt::BufferAddress, + base_vertex: i32, + first_instance: u32, + instance_count: u32, + first_instance_location: Option, + }, + DrawIndirect { + topology: u32, + indirect_buf: glow::Buffer, + indirect_offset: wgt::BufferAddress, + first_instance_location: Option, + }, + DrawIndexedIndirect { + topology: u32, + index_type: u32, + indirect_buf: glow::Buffer, + indirect_offset: wgt::BufferAddress, + first_instance_location: Option, + }, + Dispatch([u32; 3]), + DispatchIndirect { + indirect_buf: glow::Buffer, + indirect_offset: wgt::BufferAddress, + }, + ClearBuffer { + dst: Buffer, + dst_target: BindTarget, + range: crate::MemoryRange, + }, + CopyBufferToBuffer { + src: Buffer, + src_target: BindTarget, + dst: Buffer, + dst_target: BindTarget, + copy: crate::BufferCopy, + }, + #[cfg(webgl)] + CopyExternalImageToTexture { + src: wgt::CopyExternalImageSourceInfo, + dst: glow::Texture, + dst_target: BindTarget, + dst_format: wgt::TextureFormat, + dst_premultiplication: bool, + copy: crate::TextureCopy, + }, + CopyTextureToTexture { + src: glow::Texture, + src_target: BindTarget, + dst: glow::Texture, + dst_target: BindTarget, + copy: crate::TextureCopy, + }, + CopyBufferToTexture { + src: Buffer, + #[allow(unused)] + src_target: BindTarget, + dst: glow::Texture, + dst_target: BindTarget, + dst_format: wgt::TextureFormat, + copy: crate::BufferTextureCopy, + }, + CopyTextureToBuffer { + src: glow::Texture, + src_target: BindTarget, + src_format: wgt::TextureFormat, + dst: Buffer, + #[allow(unused)] + dst_target: BindTarget, + copy: crate::BufferTextureCopy, + }, + SetIndexBuffer(glow::Buffer), + BeginQuery(glow::Query, BindTarget), + EndQuery(BindTarget), + TimestampQuery(glow::Query), + CopyQueryResults { + query_range: Range, + dst: Buffer, + dst_target: BindTarget, + dst_offset: wgt::BufferAddress, + }, + ResetFramebuffer { + is_default: bool, + }, + BindAttachment { + attachment: u32, + view: TextureView, + depth_slice: Option, + sample_count: u32, + }, + ResolveAttachment { + attachment: u32, + dst: TextureView, + size: wgt::Extent3d, + }, + InvalidateAttachments(InvalidatedAttachments), + SetDrawColorBuffers(u8), + ClearColorF { + draw_buffer: u32, + color: [f32; 4], + is_srgb: bool, + }, + ClearColorU(u32, [u32; 4]), + ClearColorI(u32, [i32; 4]), + ClearDepth(f32), + ClearStencil(u32), + // Clearing both the depth and stencil buffer individually appears to + // result in the stencil buffer failing to clear, atleast in WebGL. + // It is also more efficient to emit a single command instead of two for + // this. + ClearDepthAndStencil(f32, u32), + BufferBarrier(glow::Buffer, wgt::BufferUses), + TextureBarrier(wgt::TextureUses), + SetViewport { + rect: crate::Rect, + depth: Range, + }, + SetScissor(crate::Rect), + SetStencilFunc { + face: u32, + function: u32, + reference: u32, + read_mask: u32, + }, + SetStencilOps { + face: u32, + write_mask: u32, + ops: StencilOps, + }, + SetDepth(DepthState), + SetDepthBias(wgt::DepthBiasState), + ConfigureDepthStencil(crate::FormatAspects), + SetAlphaToCoverage(bool), + SetVertexAttribute { + buffer: Option, + buffer_desc: VertexBufferDesc, + attribute_desc: AttributeDesc, + }, + UnsetVertexAttribute(u32), + SetVertexBuffer { + index: u32, + buffer: BufferBinding, + buffer_desc: VertexBufferDesc, + }, + SetProgram(glow::Program), + SetPrimitive(PrimitiveState), + SetBlendConstant([f32; 4]), + SetColorTarget { + draw_buffer_index: Option, + desc: ColorTargetDesc, + }, + BindBuffer { + target: BindTarget, + slot: u32, + buffer: glow::Buffer, + offset: i32, + size: i32, + }, + BindSampler(u32, Option), + BindTexture { + slot: u32, + texture: glow::Texture, + target: BindTarget, + aspects: crate::FormatAspects, + mip_levels: Range, + }, + BindImage { + slot: u32, + binding: ImageBinding, + }, + InsertDebugMarker(Range), + PushDebugGroup(Range), + PopDebugGroup, + SetImmediates { + uniform: ImmediateDesc, + /// Offset from the start of the `data_bytes` + offset: u32, + }, + SetClipDistances { + old_count: u32, + new_count: u32, + }, +} + +#[derive(Default)] +pub struct CommandBuffer { + label: Option, + commands: Vec, + data_bytes: Vec, + queries: Vec, +} + +impl crate::DynCommandBuffer for CommandBuffer {} + +impl fmt::Debug for CommandBuffer { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + let mut builder = f.debug_struct("CommandBuffer"); + if let Some(ref label) = self.label { + builder.field("label", label); + } + builder.finish() + } +} + +#[cfg(send_sync)] +unsafe impl Sync for CommandBuffer {} +#[cfg(send_sync)] +unsafe impl Send for CommandBuffer {} + +//TODO: we would have something like `Arc` +// here and in the command buffers. So that everything grows +// inside the encoder and stays there until `reset_all`. + +pub struct CommandEncoder { + cmd_buffer: CommandBuffer, + state: command::State, + private_caps: PrivateCapabilities, + counters: Arc, +} + +impl fmt::Debug for CommandEncoder { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("CommandEncoder") + .field("cmd_buffer", &self.cmd_buffer) + .finish() + } +} + +#[cfg(send_sync)] +unsafe impl Sync for CommandEncoder {} +#[cfg(send_sync)] +unsafe impl Send for CommandEncoder {} + +#[cfg(not(webgl))] +fn gl_debug_message_callback(source: u32, gltype: u32, id: u32, severity: u32, message: &str) { + let source_str = match source { + glow::DEBUG_SOURCE_API => "API", + glow::DEBUG_SOURCE_WINDOW_SYSTEM => "Window System", + glow::DEBUG_SOURCE_SHADER_COMPILER => "ShaderCompiler", + glow::DEBUG_SOURCE_THIRD_PARTY => "Third Party", + glow::DEBUG_SOURCE_APPLICATION => "Application", + glow::DEBUG_SOURCE_OTHER => "Other", + _ => unreachable!(), + }; + + let log_severity = match severity { + glow::DEBUG_SEVERITY_HIGH => log::Level::Error, + glow::DEBUG_SEVERITY_MEDIUM => log::Level::Warn, + glow::DEBUG_SEVERITY_LOW => log::Level::Debug, + glow::DEBUG_SEVERITY_NOTIFICATION => log::Level::Trace, + _ => unreachable!(), + }; + + let type_str = match gltype { + glow::DEBUG_TYPE_DEPRECATED_BEHAVIOR => "Deprecated Behavior", + glow::DEBUG_TYPE_ERROR => "Error", + glow::DEBUG_TYPE_MARKER => "Marker", + glow::DEBUG_TYPE_OTHER => "Other", + glow::DEBUG_TYPE_PERFORMANCE => "Performance", + glow::DEBUG_TYPE_POP_GROUP => "Pop Group", + glow::DEBUG_TYPE_PORTABILITY => "Portability", + glow::DEBUG_TYPE_PUSH_GROUP => "Push Group", + glow::DEBUG_TYPE_UNDEFINED_BEHAVIOR => "Undefined Behavior", + _ => unreachable!(), + }; + + let _ = std::panic::catch_unwind(|| { + log::log!( + log_severity, + "GLES: [{source_str}/{type_str}] ID {id} : {message}" + ); + }); + + #[cfg(feature = "validation_canary")] + if cfg!(debug_assertions) && log_severity == log::Level::Error { + // Set canary and continue + crate::VALIDATION_CANARY.add(message.to_string()); + } +} + +// If we are using `std`, then use `Mutex` to provide `Send` and `Sync` +cfg_if::cfg_if! { + if #[cfg(gles_with_std)] { + type MaybeMutex = std::sync::Mutex; + + fn lock(mutex: &MaybeMutex) -> std::sync::MutexGuard<'_, T> { + mutex.lock().unwrap() + } + } else { + // It should be impossible for any build configuration to trigger this error + // It is intended only as a guard against changes elsewhere causing the use of + // `RefCell` here to become unsound. + #[cfg(all(send_sync, not(feature = "fragile-send-sync-non-atomic-wasm")))] + compile_error!("cannot provide non-fragile Send+Sync without std"); + + type MaybeMutex = core::cell::RefCell; + + fn lock(mutex: &MaybeMutex) -> core::cell::RefMut<'_, T> { + mutex.borrow_mut() + } + } +} diff --git a/third_party/wgpu-hal-29.0.4/src/gles/queue.rs b/third_party/wgpu-hal-29.0.4/src/gles/queue.rs new file mode 100644 index 0000000..a1f09a5 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/queue.rs @@ -0,0 +1,1963 @@ +use alloc::sync::Arc; +use alloc::vec; +use core::sync::atomic::Ordering; + +use arrayvec::ArrayVec; +use glow::HasContext; + +use super::{conv::is_layered_target, lock, Command as C, PrivateCapabilities}; + +const DEBUG_ID: u32 = 0; + +fn extract_marker<'a>(data: &'a [u8], range: &core::ops::Range) -> &'a str { + core::str::from_utf8(&data[range.start as usize..range.end as usize]).unwrap() +} + +fn to_debug_str(s: &str) -> &str { + // The spec mentions that if the length given to debug functions is negative, + // the implementation will access the ptr and look for a null that terminates + // the string but some implementations will try to access the ptr even if the + // length is 0. + if s.is_empty() { + "" + } else { + s + } +} + +fn get_2d_target(target: u32, array_layer: u32) -> u32 { + const CUBEMAP_FACES: [u32; 6] = [ + glow::TEXTURE_CUBE_MAP_POSITIVE_X, + glow::TEXTURE_CUBE_MAP_NEGATIVE_X, + glow::TEXTURE_CUBE_MAP_POSITIVE_Y, + glow::TEXTURE_CUBE_MAP_NEGATIVE_Y, + glow::TEXTURE_CUBE_MAP_POSITIVE_Z, + glow::TEXTURE_CUBE_MAP_NEGATIVE_Z, + ]; + + match target { + glow::TEXTURE_2D => target, + glow::TEXTURE_CUBE_MAP => CUBEMAP_FACES[array_layer as usize], + _ => unreachable!(), + } +} + +fn get_z_offset(target: u32, base: &crate::TextureCopyBase) -> u32 { + match target { + glow::TEXTURE_2D_ARRAY | glow::TEXTURE_CUBE_MAP_ARRAY => base.array_layer, + glow::TEXTURE_3D => base.origin.z, + _ => unreachable!(), + } +} + +impl super::Queue { + /// Performs a manual shader clear, used as a workaround for a clearing bug on mesa + unsafe fn perform_shader_clear(&self, gl: &glow::Context, draw_buffer: u32, color: [f32; 4]) { + let shader_clear = self + .shader_clear_program + .as_ref() + .expect("shader_clear_program should always be set if the workaround is enabled"); + unsafe { gl.use_program(Some(shader_clear.program)) }; + unsafe { + gl.uniform_4_f32( + Some(&shader_clear.color_uniform_location), + color[0], + color[1], + color[2], + color[3], + ) + }; + unsafe { gl.disable(glow::DEPTH_TEST) }; + unsafe { gl.disable(glow::STENCIL_TEST) }; + unsafe { gl.disable(glow::SCISSOR_TEST) }; + unsafe { gl.disable(glow::BLEND) }; + unsafe { gl.disable(glow::CULL_FACE) }; + unsafe { gl.draw_buffers(&[glow::COLOR_ATTACHMENT0 + draw_buffer]) }; + unsafe { gl.draw_arrays(glow::TRIANGLES, 0, 3) }; + + let draw_buffer_count = self.draw_buffer_count.load(Ordering::Relaxed); + if draw_buffer_count != 0 { + // Reset the draw buffers to what they were before the clear + let indices = (0..draw_buffer_count as u32) + .map(|i| glow::COLOR_ATTACHMENT0 + i) + .collect::>(); + unsafe { gl.draw_buffers(&indices) }; + } + } + + unsafe fn reset_state(&self, gl: &glow::Context) { + unsafe { gl.use_program(None) }; + unsafe { gl.bind_framebuffer(glow::FRAMEBUFFER, None) }; + unsafe { gl.disable(glow::DEPTH_TEST) }; + unsafe { gl.disable(glow::STENCIL_TEST) }; + unsafe { gl.disable(glow::SCISSOR_TEST) }; + unsafe { gl.disable(glow::BLEND) }; + unsafe { gl.disable(glow::CULL_FACE) }; + unsafe { gl.disable(glow::POLYGON_OFFSET_FILL) }; + unsafe { gl.disable(glow::SAMPLE_ALPHA_TO_COVERAGE) }; + if self.features.contains(wgt::Features::DEPTH_CLIP_CONTROL) { + unsafe { gl.disable(glow::DEPTH_CLAMP) }; + } + + unsafe { gl.bind_buffer(glow::ELEMENT_ARRAY_BUFFER, None) }; + let mut current_index_buffer = self.current_index_buffer.lock(); + *current_index_buffer = None; + } + + unsafe fn set_attachment( + &self, + gl: &glow::Context, + fbo_target: u32, + attachment: u32, + view: &super::TextureView, + depth_slice: Option, + sample_count: u32, + ) { + match view.inner { + super::TextureInner::Renderbuffer { raw } => { + unsafe { + gl.framebuffer_renderbuffer( + fbo_target, + attachment, + glow::RENDERBUFFER, + Some(raw), + ) + }; + } + super::TextureInner::DefaultRenderbuffer => panic!("Unexpected default RBO"), + super::TextureInner::Texture { raw, target } => { + let num_layers = view.array_layers.end - view.array_layers.start; + if num_layers > 1 { + #[cfg(webgl)] + unsafe { + gl.framebuffer_texture_multiview_ovr( + fbo_target, + attachment, + Some(raw), + view.mip_levels.start as i32, + view.array_layers.start as i32, + num_layers as i32, + ) + }; + } else if is_layered_target(target) { + let layer = if target == glow::TEXTURE_3D { + depth_slice.unwrap() as i32 + } else { + view.array_layers.start as i32 + }; + unsafe { + gl.framebuffer_texture_layer( + fbo_target, + attachment, + Some(raw), + view.mip_levels.start as i32, + layer, + ) + }; + } else { + unsafe { + assert_eq!(view.mip_levels.len(), 1); + if sample_count != 1 { + gl.framebuffer_texture_2d_multisample( + fbo_target, + attachment, + get_2d_target(target, view.array_layers.start), + Some(raw), + view.mip_levels.start as i32, + sample_count as i32, + ) + } else { + gl.framebuffer_texture_2d( + fbo_target, + attachment, + get_2d_target(target, view.array_layers.start), + Some(raw), + view.mip_levels.start as i32, + ) + } + }; + } + } + #[cfg(webgl)] + super::TextureInner::ExternalFramebuffer { ref inner } => unsafe { + gl.bind_external_framebuffer(glow::FRAMEBUFFER, inner); + }, + #[cfg(native)] + super::TextureInner::ExternalNativeFramebuffer { ref inner } => unsafe { + gl.bind_framebuffer(glow::FRAMEBUFFER, Some(*inner)); + }, + } + } + + unsafe fn process( + &self, + gl: &glow::Context, + command: &C, + #[cfg_attr(target_arch = "wasm32", allow(unused))] data_bytes: &[u8], + queries: &[glow::Query], + ) { + match *command { + C::Draw { + topology, + first_vertex, + vertex_count, + instance_count, + first_instance, + ref first_instance_location, + } => { + let supports_full_instancing = self + .shared + .private_caps + .contains(PrivateCapabilities::FULLY_FEATURED_INSTANCING); + + if supports_full_instancing { + unsafe { + gl.draw_arrays_instanced_base_instance( + topology, + first_vertex as i32, + vertex_count as i32, + instance_count as i32, + first_instance, + ) + } + } else { + unsafe { + gl.uniform_1_u32(first_instance_location.as_ref(), first_instance); + } + + // Don't use `gl.draw_arrays` for `instance_count == 1`. + // Angle has a bug where it doesn't consider the instance divisor when `DYNAMIC_DRAW` is used in `draw_arrays`. + // See https://github.com/gfx-rs/wgpu/issues/3578 + unsafe { + gl.draw_arrays_instanced( + topology, + first_vertex as i32, + vertex_count as i32, + instance_count as i32, + ) + } + }; + } + C::DrawIndexed { + topology, + index_type, + index_count, + index_offset, + base_vertex, + first_instance, + instance_count, + ref first_instance_location, + } => { + let supports_full_instancing = self + .shared + .private_caps + .contains(PrivateCapabilities::FULLY_FEATURED_INSTANCING); + + if supports_full_instancing { + unsafe { + gl.draw_elements_instanced_base_vertex_base_instance( + topology, + index_count as i32, + index_type, + index_offset as i32, + instance_count as i32, + base_vertex, + first_instance, + ) + } + } else { + unsafe { gl.uniform_1_u32(first_instance_location.as_ref(), first_instance) }; + + if base_vertex == 0 { + unsafe { + // Don't use `gl.draw_elements`/`gl.draw_elements_base_vertex` for `instance_count == 1`. + // Angle has a bug where it doesn't consider the instance divisor when `DYNAMIC_DRAW` is used in `gl.draw_elements`/`gl.draw_elements_base_vertex`. + // See https://github.com/gfx-rs/wgpu/issues/3578 + gl.draw_elements_instanced( + topology, + index_count as i32, + index_type, + index_offset as i32, + instance_count as i32, + ) + } + } else { + // If we've gotten here, wgpu-core has already validated that this function exists via the DownlevelFlags::BASE_VERTEX feature. + unsafe { + gl.draw_elements_instanced_base_vertex( + topology, + index_count as _, + index_type, + index_offset as i32, + instance_count as i32, + base_vertex, + ) + } + } + } + } + C::DrawIndirect { + topology, + indirect_buf, + indirect_offset, + ref first_instance_location, + } => { + unsafe { gl.uniform_1_u32(first_instance_location.as_ref(), 0) }; + + unsafe { gl.bind_buffer(glow::DRAW_INDIRECT_BUFFER, Some(indirect_buf)) }; + unsafe { gl.draw_arrays_indirect_offset(topology, indirect_offset as i32) }; + } + C::DrawIndexedIndirect { + topology, + index_type, + indirect_buf, + indirect_offset, + ref first_instance_location, + } => { + unsafe { gl.uniform_1_u32(first_instance_location.as_ref(), 0) }; + + unsafe { gl.bind_buffer(glow::DRAW_INDIRECT_BUFFER, Some(indirect_buf)) }; + unsafe { + gl.draw_elements_indirect_offset(topology, index_type, indirect_offset as i32) + }; + } + C::Dispatch(group_counts) => { + unsafe { gl.dispatch_compute(group_counts[0], group_counts[1], group_counts[2]) }; + } + C::DispatchIndirect { + indirect_buf, + indirect_offset, + } => { + unsafe { gl.bind_buffer(glow::DISPATCH_INDIRECT_BUFFER, Some(indirect_buf)) }; + unsafe { gl.dispatch_compute_indirect(indirect_offset as i32) }; + } + C::ClearBuffer { + ref dst, + dst_target, + ref range, + } => match dst.raw { + Some(buffer) => { + // When `INDEX_BUFFER_ROLE_CHANGE` isn't available, we can't copy into the + // index buffer from the zero buffer. This would fail in Chrome with the + // following message: + // + // > Cannot copy into an element buffer destination from a non-element buffer + // > source + // + // Instead, we'll upload zeroes into the buffer. + let can_use_zero_buffer = self + .shared + .private_caps + .contains(PrivateCapabilities::INDEX_BUFFER_ROLE_CHANGE) + || dst_target != glow::ELEMENT_ARRAY_BUFFER; + + if can_use_zero_buffer { + unsafe { gl.bind_buffer(glow::COPY_READ_BUFFER, Some(self.zero_buffer)) }; + unsafe { gl.bind_buffer(dst_target, Some(buffer)) }; + let mut dst_offset = range.start; + while dst_offset < range.end { + let size = (range.end - dst_offset).min(super::ZERO_BUFFER_SIZE as u64); + unsafe { + gl.copy_buffer_sub_data( + glow::COPY_READ_BUFFER, + dst_target, + 0, + dst_offset as i32, + size as i32, + ) + }; + dst_offset += size; + } + } else { + unsafe { gl.bind_buffer(dst_target, Some(buffer)) }; + let zeroes = vec![0u8; (range.end - range.start) as usize]; + unsafe { + gl.buffer_sub_data_u8_slice(dst_target, range.start as i32, &zeroes) + }; + } + } + None => { + lock(dst.data.as_ref().unwrap()).as_mut_slice() + [range.start as usize..range.end as usize] + .fill(0); + } + }, + C::CopyBufferToBuffer { + ref src, + src_target, + ref dst, + dst_target, + copy, + } => { + let copy_src_target = glow::COPY_READ_BUFFER; + let is_index_buffer_only_element_dst = !self + .shared + .private_caps + .contains(PrivateCapabilities::INDEX_BUFFER_ROLE_CHANGE) + && dst_target == glow::ELEMENT_ARRAY_BUFFER + || src_target == glow::ELEMENT_ARRAY_BUFFER; + + // WebGL not allowed to copy data from other targets to element buffer and can't copy element data to other buffers + let copy_dst_target = if is_index_buffer_only_element_dst { + glow::ELEMENT_ARRAY_BUFFER + } else { + glow::COPY_WRITE_BUFFER + }; + let size = copy.size.get() as usize; + match (src.raw, dst.raw) { + (Some(ref src), Some(ref dst)) => { + unsafe { gl.bind_buffer(copy_src_target, Some(*src)) }; + unsafe { gl.bind_buffer(copy_dst_target, Some(*dst)) }; + unsafe { + gl.copy_buffer_sub_data( + copy_src_target, + copy_dst_target, + copy.src_offset as _, + copy.dst_offset as _, + copy.size.get() as _, + ) + }; + } + (Some(src), None) => { + let mut data = lock(dst.data.as_ref().unwrap()); + let dst_data = &mut data.as_mut_slice() + [copy.dst_offset as usize..copy.dst_offset as usize + size]; + + unsafe { gl.bind_buffer(copy_src_target, Some(src)) }; + unsafe { + self.shared.get_buffer_sub_data( + gl, + copy_src_target, + copy.src_offset as i32, + dst_data, + ) + }; + } + (None, Some(dst)) => { + let data = lock(src.data.as_ref().unwrap()); + let src_data = &data.as_slice() + [copy.src_offset as usize..copy.src_offset as usize + size]; + unsafe { gl.bind_buffer(copy_dst_target, Some(dst)) }; + unsafe { + gl.buffer_sub_data_u8_slice( + copy_dst_target, + copy.dst_offset as i32, + src_data, + ) + }; + } + (None, None) => { + todo!() + } + } + unsafe { gl.bind_buffer(copy_src_target, None) }; + if is_index_buffer_only_element_dst { + unsafe { + gl.bind_buffer( + glow::ELEMENT_ARRAY_BUFFER, + *self.current_index_buffer.lock(), + ) + }; + } else { + unsafe { gl.bind_buffer(copy_dst_target, None) }; + } + } + #[cfg(webgl)] + C::CopyExternalImageToTexture { + ref src, + dst, + dst_target, + dst_format, + dst_premultiplication, + ref copy, + } => { + const UNPACK_FLIP_Y_WEBGL: u32 = + web_sys::WebGl2RenderingContext::UNPACK_FLIP_Y_WEBGL; + const UNPACK_PREMULTIPLY_ALPHA_WEBGL: u32 = + web_sys::WebGl2RenderingContext::UNPACK_PREMULTIPLY_ALPHA_WEBGL; + + unsafe { + if src.flip_y { + gl.pixel_store_bool(UNPACK_FLIP_Y_WEBGL, true); + } + if dst_premultiplication { + gl.pixel_store_bool(UNPACK_PREMULTIPLY_ALPHA_WEBGL, true); + } + } + + unsafe { gl.bind_texture(dst_target, Some(dst)) }; + let format_desc = self.shared.describe_texture_format(dst_format); + if is_layered_target(dst_target) { + let z_offset = get_z_offset(dst_target, ©.dst_base); + + match src.source { + wgt::ExternalImageSource::ImageBitmap(ref b) => unsafe { + gl.tex_sub_image_3d_with_image_bitmap( + dst_target, + copy.dst_base.mip_level as i32, + copy.dst_base.origin.x as i32, + copy.dst_base.origin.y as i32, + z_offset as i32, + copy.size.width as i32, + copy.size.height as i32, + copy.size.depth as i32, + format_desc.external, + format_desc.data_type, + b, + ); + }, + wgt::ExternalImageSource::HTMLImageElement(ref i) => unsafe { + gl.tex_sub_image_3d_with_html_image_element( + dst_target, + copy.dst_base.mip_level as i32, + copy.dst_base.origin.x as i32, + copy.dst_base.origin.y as i32, + z_offset as i32, + copy.size.width as i32, + copy.size.height as i32, + copy.size.depth as i32, + format_desc.external, + format_desc.data_type, + i, + ); + }, + wgt::ExternalImageSource::HTMLVideoElement(ref v) => unsafe { + gl.tex_sub_image_3d_with_html_video_element( + dst_target, + copy.dst_base.mip_level as i32, + copy.dst_base.origin.x as i32, + copy.dst_base.origin.y as i32, + z_offset as i32, + copy.size.width as i32, + copy.size.height as i32, + copy.size.depth as i32, + format_desc.external, + format_desc.data_type, + v, + ); + }, + #[cfg(web_sys_unstable_apis)] + wgt::ExternalImageSource::VideoFrame(ref v) => unsafe { + gl.tex_sub_image_3d_with_video_frame( + dst_target, + copy.dst_base.mip_level as i32, + copy.dst_base.origin.x as i32, + copy.dst_base.origin.y as i32, + z_offset as i32, + copy.size.width as i32, + copy.size.height as i32, + copy.size.depth as i32, + format_desc.external, + format_desc.data_type, + v, + ) + }, + wgt::ExternalImageSource::ImageData(ref i) => unsafe { + gl.tex_sub_image_3d_with_image_data( + dst_target, + copy.dst_base.mip_level as i32, + copy.dst_base.origin.x as i32, + copy.dst_base.origin.y as i32, + z_offset as i32, + copy.size.width as i32, + copy.size.height as i32, + copy.size.depth as i32, + format_desc.external, + format_desc.data_type, + i, + ); + }, + wgt::ExternalImageSource::HTMLCanvasElement(ref c) => unsafe { + gl.tex_sub_image_3d_with_html_canvas_element( + dst_target, + copy.dst_base.mip_level as i32, + copy.dst_base.origin.x as i32, + copy.dst_base.origin.y as i32, + z_offset as i32, + copy.size.width as i32, + copy.size.height as i32, + copy.size.depth as i32, + format_desc.external, + format_desc.data_type, + c, + ); + }, + wgt::ExternalImageSource::OffscreenCanvas(_) => unreachable!(), + } + } else { + let dst_target = get_2d_target(dst_target, copy.dst_base.array_layer); + + match src.source { + wgt::ExternalImageSource::ImageBitmap(ref b) => unsafe { + gl.tex_sub_image_2d_with_image_bitmap_and_width_and_height( + dst_target, + copy.dst_base.mip_level as i32, + copy.dst_base.origin.x as i32, + copy.dst_base.origin.y as i32, + copy.size.width as i32, + copy.size.height as i32, + format_desc.external, + format_desc.data_type, + b, + ); + }, + wgt::ExternalImageSource::HTMLImageElement(ref i) => unsafe { + gl.tex_sub_image_2d_with_html_image_and_width_and_height( + dst_target, + copy.dst_base.mip_level as i32, + copy.dst_base.origin.x as i32, + copy.dst_base.origin.y as i32, + copy.size.width as i32, + copy.size.height as i32, + format_desc.external, + format_desc.data_type, + i, + ) + }, + wgt::ExternalImageSource::HTMLVideoElement(ref v) => unsafe { + gl.tex_sub_image_2d_with_html_video_and_width_and_height( + dst_target, + copy.dst_base.mip_level as i32, + copy.dst_base.origin.x as i32, + copy.dst_base.origin.y as i32, + copy.size.width as i32, + copy.size.height as i32, + format_desc.external, + format_desc.data_type, + v, + ) + }, + #[cfg(web_sys_unstable_apis)] + wgt::ExternalImageSource::VideoFrame(ref v) => unsafe { + gl.tex_sub_image_2d_with_video_frame_and_width_and_height( + dst_target, + copy.dst_base.mip_level as i32, + copy.dst_base.origin.x as i32, + copy.dst_base.origin.y as i32, + copy.size.width as i32, + copy.size.height as i32, + format_desc.external, + format_desc.data_type, + v, + ) + }, + wgt::ExternalImageSource::ImageData(ref i) => unsafe { + gl.tex_sub_image_2d_with_image_data_and_width_and_height( + dst_target, + copy.dst_base.mip_level as i32, + copy.dst_base.origin.x as i32, + copy.dst_base.origin.y as i32, + copy.size.width as i32, + copy.size.height as i32, + format_desc.external, + format_desc.data_type, + i, + ); + }, + wgt::ExternalImageSource::HTMLCanvasElement(ref c) => unsafe { + gl.tex_sub_image_2d_with_html_canvas_and_width_and_height( + dst_target, + copy.dst_base.mip_level as i32, + copy.dst_base.origin.x as i32, + copy.dst_base.origin.y as i32, + copy.size.width as i32, + copy.size.height as i32, + format_desc.external, + format_desc.data_type, + c, + ) + }, + wgt::ExternalImageSource::OffscreenCanvas(_) => unreachable!(), + } + } + + unsafe { + if src.flip_y { + gl.pixel_store_bool(UNPACK_FLIP_Y_WEBGL, false); + } + if dst_premultiplication { + gl.pixel_store_bool(UNPACK_PREMULTIPLY_ALPHA_WEBGL, false); + } + } + } + C::CopyTextureToTexture { + src, + src_target, + dst, + dst_target, + ref copy, + } => { + //TODO: handle 3D copies + unsafe { gl.bind_framebuffer(glow::READ_FRAMEBUFFER, Some(self.copy_fbo)) }; + if is_layered_target(src_target) { + //TODO: handle GLES without framebuffer_texture_3d + unsafe { + gl.framebuffer_texture_layer( + glow::READ_FRAMEBUFFER, + glow::COLOR_ATTACHMENT0, + Some(src), + copy.src_base.mip_level as i32, + copy.src_base.array_layer as i32, + ) + }; + } else { + unsafe { + gl.framebuffer_texture_2d( + glow::READ_FRAMEBUFFER, + glow::COLOR_ATTACHMENT0, + src_target, + Some(src), + copy.src_base.mip_level as i32, + ) + }; + } + + unsafe { gl.bind_texture(dst_target, Some(dst)) }; + if is_layered_target(dst_target) { + unsafe { + gl.copy_tex_sub_image_3d( + dst_target, + copy.dst_base.mip_level as i32, + copy.dst_base.origin.x as i32, + copy.dst_base.origin.y as i32, + get_z_offset(dst_target, ©.dst_base) as i32, + copy.src_base.origin.x as i32, + copy.src_base.origin.y as i32, + copy.size.width as i32, + copy.size.height as i32, + ) + }; + } else { + unsafe { + gl.copy_tex_sub_image_2d( + get_2d_target(dst_target, copy.dst_base.array_layer), + copy.dst_base.mip_level as i32, + copy.dst_base.origin.x as i32, + copy.dst_base.origin.y as i32, + copy.src_base.origin.x as i32, + copy.src_base.origin.y as i32, + copy.size.width as i32, + copy.size.height as i32, + ) + }; + } + } + C::CopyBufferToTexture { + ref src, + src_target: _, + dst, + dst_target, + dst_format, + ref copy, + } => { + let (block_width, block_height) = dst_format.block_dimensions(); + let block_size = dst_format.block_copy_size(None).unwrap(); + let format_desc = self.shared.describe_texture_format(dst_format); + let row_texels = copy + .buffer_layout + .bytes_per_row + .map_or(0, |bpr| block_width * bpr / block_size); + let column_texels = copy + .buffer_layout + .rows_per_image + .map_or(0, |rpi| block_height * rpi); + + unsafe { gl.bind_texture(dst_target, Some(dst)) }; + unsafe { gl.pixel_store_i32(glow::UNPACK_ROW_LENGTH, row_texels as i32) }; + unsafe { gl.pixel_store_i32(glow::UNPACK_IMAGE_HEIGHT, column_texels as i32) }; + let mut unbind_unpack_buffer = false; + if !dst_format.is_compressed() { + let buffer_data; + let unpack_data = match src.raw { + Some(buffer) => { + unsafe { gl.bind_buffer(glow::PIXEL_UNPACK_BUFFER, Some(buffer)) }; + unbind_unpack_buffer = true; + glow::PixelUnpackData::BufferOffset(copy.buffer_layout.offset as u32) + } + None => { + buffer_data = lock(src.data.as_ref().unwrap()); + let src_data = + &buffer_data.as_slice()[copy.buffer_layout.offset as usize..]; + glow::PixelUnpackData::Slice(Some(src_data)) + } + }; + if is_layered_target(dst_target) { + unsafe { + gl.tex_sub_image_3d( + dst_target, + copy.texture_base.mip_level as i32, + copy.texture_base.origin.x as i32, + copy.texture_base.origin.y as i32, + get_z_offset(dst_target, ©.texture_base) as i32, + copy.size.width as i32, + copy.size.height as i32, + copy.size.depth as i32, + format_desc.external, + format_desc.data_type, + unpack_data, + ) + }; + } else { + unsafe { + gl.tex_sub_image_2d( + get_2d_target(dst_target, copy.texture_base.array_layer), + copy.texture_base.mip_level as i32, + copy.texture_base.origin.x as i32, + copy.texture_base.origin.y as i32, + copy.size.width as i32, + copy.size.height as i32, + format_desc.external, + format_desc.data_type, + unpack_data, + ) + }; + } + } else { + let bytes_per_row = copy + .buffer_layout + .bytes_per_row + .unwrap_or(copy.size.width * block_size); + let minimum_rows_per_image = copy.size.height.div_ceil(block_height); + let rows_per_image = copy + .buffer_layout + .rows_per_image + .unwrap_or(minimum_rows_per_image); + + let bytes_per_image = bytes_per_row * rows_per_image; + let minimum_bytes_per_image = bytes_per_row * minimum_rows_per_image; + let bytes_in_upload = + (bytes_per_image * (copy.size.depth - 1)) + minimum_bytes_per_image; + let offset = copy.buffer_layout.offset as u32; + + let buffer_data; + let unpack_data = match src.raw { + Some(buffer) => { + unsafe { gl.bind_buffer(glow::PIXEL_UNPACK_BUFFER, Some(buffer)) }; + unbind_unpack_buffer = true; + glow::CompressedPixelUnpackData::BufferRange( + offset..offset + bytes_in_upload, + ) + } + None => { + buffer_data = lock(src.data.as_ref().unwrap()); + let src_data = &buffer_data.as_slice() + [(offset as usize)..(offset + bytes_in_upload) as usize]; + glow::CompressedPixelUnpackData::Slice(src_data) + } + }; + + if is_layered_target(dst_target) { + unsafe { + gl.compressed_tex_sub_image_3d( + dst_target, + copy.texture_base.mip_level as i32, + copy.texture_base.origin.x as i32, + copy.texture_base.origin.y as i32, + get_z_offset(dst_target, ©.texture_base) as i32, + copy.size.width as i32, + copy.size.height as i32, + copy.size.depth as i32, + format_desc.internal, + unpack_data, + ) + }; + } else { + unsafe { + gl.compressed_tex_sub_image_2d( + get_2d_target(dst_target, copy.texture_base.array_layer), + copy.texture_base.mip_level as i32, + copy.texture_base.origin.x as i32, + copy.texture_base.origin.y as i32, + copy.size.width as i32, + copy.size.height as i32, + format_desc.internal, + unpack_data, + ) + }; + } + } + if unbind_unpack_buffer { + unsafe { gl.bind_buffer(glow::PIXEL_UNPACK_BUFFER, None) }; + } + } + C::CopyTextureToBuffer { + src, + src_target, + src_format, + ref dst, + dst_target: _, + ref copy, + } => { + let block_size = src_format.block_copy_size(None).unwrap(); + if src_format.is_compressed() { + log::error!("Not implemented yet: compressed texture copy to buffer"); + return; + } + if src_target == glow::TEXTURE_CUBE_MAP + || src_target == glow::TEXTURE_CUBE_MAP_ARRAY + { + log::error!("Not implemented yet: cubemap texture copy to buffer"); + return; + } + let format_desc = self.shared.describe_texture_format(src_format); + let row_texels = copy + .buffer_layout + .bytes_per_row + .map_or(copy.size.width, |bpr| bpr / block_size); + let column_texels = copy + .buffer_layout + .rows_per_image + .unwrap_or(copy.size.height); + + unsafe { gl.bind_framebuffer(glow::READ_FRAMEBUFFER, Some(self.copy_fbo)) }; + + let read_pixels = |offset| { + let mut buffer_data; + let unpack_data = match dst.raw { + Some(buffer) => { + unsafe { gl.pixel_store_i32(glow::PACK_ROW_LENGTH, row_texels as i32) }; + unsafe { gl.bind_buffer(glow::PIXEL_PACK_BUFFER, Some(buffer)) }; + glow::PixelPackData::BufferOffset(offset as u32) + } + None => { + buffer_data = lock(dst.data.as_ref().unwrap()); + let dst_data = &mut buffer_data.as_mut_slice()[offset as usize..]; + glow::PixelPackData::Slice(Some(dst_data)) + } + }; + unsafe { + gl.read_pixels( + copy.texture_base.origin.x as i32, + copy.texture_base.origin.y as i32, + copy.size.width as i32, + copy.size.height as i32, + format_desc.external, + format_desc.data_type, + unpack_data, + ) + }; + }; + + match src_target { + glow::TEXTURE_2D => { + unsafe { + gl.framebuffer_texture_2d( + glow::READ_FRAMEBUFFER, + glow::COLOR_ATTACHMENT0, + src_target, + Some(src), + copy.texture_base.mip_level as i32, + ) + }; + read_pixels(copy.buffer_layout.offset); + } + glow::TEXTURE_2D_ARRAY => { + unsafe { + gl.framebuffer_texture_layer( + glow::READ_FRAMEBUFFER, + glow::COLOR_ATTACHMENT0, + Some(src), + copy.texture_base.mip_level as i32, + copy.texture_base.array_layer as i32, + ) + }; + read_pixels(copy.buffer_layout.offset); + } + glow::TEXTURE_3D => { + for z in copy.texture_base.origin.z..copy.size.depth { + unsafe { + gl.framebuffer_texture_layer( + glow::READ_FRAMEBUFFER, + glow::COLOR_ATTACHMENT0, + Some(src), + copy.texture_base.mip_level as i32, + z as i32, + ) + }; + let offset = copy.buffer_layout.offset + + (z * block_size * row_texels * column_texels) as u64; + read_pixels(offset); + } + } + glow::TEXTURE_CUBE_MAP | glow::TEXTURE_CUBE_MAP_ARRAY => unimplemented!(), + _ => unreachable!(), + } + } + C::SetIndexBuffer(buffer) => { + unsafe { gl.bind_buffer(glow::ELEMENT_ARRAY_BUFFER, Some(buffer)) }; + let mut current_index_buffer = self.current_index_buffer.lock(); + *current_index_buffer = Some(buffer); + } + C::BeginQuery(query, target) => { + unsafe { gl.begin_query(target, query) }; + } + C::EndQuery(target) => { + unsafe { gl.end_query(target) }; + } + C::TimestampQuery(query) => { + unsafe { gl.query_counter(query, glow::TIMESTAMP) }; + } + C::CopyQueryResults { + ref query_range, + ref dst, + dst_target, + dst_offset, + } => { + if self + .shared + .private_caps + .contains(PrivateCapabilities::QUERY_BUFFERS) + && dst.raw.is_some() + { + unsafe { + // We're assuming that the only relevant queries are 8 byte timestamps or + // occlusion tests. + let query_size = 8; + + let query_range_size = query_size * query_range.len(); + + let buffer = gl.create_buffer().ok(); + gl.bind_buffer(glow::QUERY_BUFFER, buffer); + gl.buffer_data_size( + glow::QUERY_BUFFER, + query_range_size as _, + glow::STREAM_COPY, + ); + + for (i, &query) in queries + [query_range.start as usize..query_range.end as usize] + .iter() + .enumerate() + { + gl.get_query_parameter_u64_with_offset( + query, + glow::QUERY_RESULT, + query_size * i, + ) + } + gl.bind_buffer(dst_target, dst.raw); + gl.copy_buffer_sub_data( + glow::QUERY_BUFFER, + dst_target, + 0, + dst_offset as _, + query_range_size as _, + ); + if let Some(buffer) = buffer { + gl.delete_buffer(buffer) + } + } + } else { + let mut temp_query_results = self.temp_query_results.lock(); + temp_query_results.clear(); + for &query in + queries[query_range.start as usize..query_range.end as usize].iter() + { + let mut result: u64 = 0; + unsafe { + if self + .shared + .private_caps + .contains(PrivateCapabilities::QUERY_64BIT) + { + let result: *mut u64 = &mut result; + gl.get_query_parameter_u64_with_offset( + query, + glow::QUERY_RESULT, + result as usize, + ) + } else { + result = + gl.get_query_parameter_u32(query, glow::QUERY_RESULT) as u64; + } + }; + temp_query_results.push(result); + } + let query_data = bytemuck::cast_slice(&temp_query_results); + match dst.raw { + Some(buffer) => { + unsafe { gl.bind_buffer(dst_target, Some(buffer)) }; + unsafe { + gl.buffer_sub_data_u8_slice( + dst_target, + dst_offset as i32, + query_data, + ) + }; + } + None => { + let data = &mut lock(dst.data.as_ref().unwrap()); + let len = query_data.len().min(data.len()); + data[..len].copy_from_slice(&query_data[..len]); + } + } + } + } + C::ResetFramebuffer { is_default } => { + if is_default { + unsafe { gl.bind_framebuffer(glow::DRAW_FRAMEBUFFER, None) }; + } else { + unsafe { gl.bind_framebuffer(glow::DRAW_FRAMEBUFFER, Some(self.draw_fbo)) }; + unsafe { + gl.framebuffer_texture_2d( + glow::DRAW_FRAMEBUFFER, + glow::DEPTH_STENCIL_ATTACHMENT, + glow::TEXTURE_2D, + None, + 0, + ) + }; + for i in 0..self.shared.limits.max_color_attachments { + let target = glow::COLOR_ATTACHMENT0 + i; + unsafe { + gl.framebuffer_texture_2d( + glow::DRAW_FRAMEBUFFER, + target, + glow::TEXTURE_2D, + None, + 0, + ) + }; + } + } + unsafe { gl.color_mask(true, true, true, true) }; + unsafe { gl.depth_mask(true) }; + unsafe { gl.stencil_mask(!0) }; + unsafe { gl.disable(glow::DEPTH_TEST) }; + unsafe { gl.disable(glow::STENCIL_TEST) }; + unsafe { gl.disable(glow::SCISSOR_TEST) }; + } + C::BindAttachment { + attachment, + ref view, + depth_slice, + sample_count, + } => { + unsafe { + self.set_attachment( + gl, + glow::DRAW_FRAMEBUFFER, + attachment, + view, + depth_slice, + sample_count, + ) + }; + } + C::ResolveAttachment { + attachment, + ref dst, + ref size, + } => { + unsafe { gl.bind_framebuffer(glow::READ_FRAMEBUFFER, Some(self.draw_fbo)) }; + unsafe { gl.read_buffer(attachment) }; + unsafe { gl.bind_framebuffer(glow::DRAW_FRAMEBUFFER, Some(self.copy_fbo)) }; + unsafe { + self.set_attachment( + gl, + glow::DRAW_FRAMEBUFFER, + glow::COLOR_ATTACHMENT0, + dst, + None, + 1, + ) + }; + unsafe { + gl.blit_framebuffer( + 0, + 0, + size.width as i32, + size.height as i32, + 0, + 0, + size.width as i32, + size.height as i32, + glow::COLOR_BUFFER_BIT, + glow::NEAREST, + ) + }; + unsafe { gl.bind_framebuffer(glow::READ_FRAMEBUFFER, None) }; + unsafe { gl.bind_framebuffer(glow::DRAW_FRAMEBUFFER, Some(self.draw_fbo)) }; + } + C::InvalidateAttachments(ref list) => { + if self + .shared + .private_caps + .contains(PrivateCapabilities::INVALIDATE_FRAMEBUFFER) + { + unsafe { gl.invalidate_framebuffer(glow::DRAW_FRAMEBUFFER, list) }; + } + } + C::SetDrawColorBuffers(count) => { + self.draw_buffer_count.store(count, Ordering::Relaxed); + let indices = (0..count as u32) + .map(|i| glow::COLOR_ATTACHMENT0 + i) + .collect::>(); + unsafe { gl.draw_buffers(&indices) }; + } + C::ClearColorF { + draw_buffer, + ref color, + is_srgb, + } => { + if self + .shared + .workarounds + .contains(super::Workarounds::MESA_I915_SRGB_SHADER_CLEAR) + && is_srgb + { + unsafe { self.perform_shader_clear(gl, draw_buffer, *color) }; + } else { + unsafe { gl.clear_buffer_f32_slice(glow::COLOR, draw_buffer, color) }; + } + } + C::ClearColorU(draw_buffer, ref color) => { + unsafe { gl.clear_buffer_u32_slice(glow::COLOR, draw_buffer, color) }; + } + C::ClearColorI(draw_buffer, ref color) => { + unsafe { gl.clear_buffer_i32_slice(glow::COLOR, draw_buffer, color) }; + } + C::ClearDepth(depth) => { + // Prefer `clear` as `clear_buffer` functions have issues on Sandy Bridge + // on Windows. + unsafe { + gl.clear_depth_f32(depth); + gl.clear(glow::DEPTH_BUFFER_BIT); + } + } + C::ClearStencil(value) => { + // Prefer `clear` as `clear_buffer` functions have issues on Sandy Bridge + // on Windows. + unsafe { + gl.clear_stencil(value as i32); + gl.clear(glow::STENCIL_BUFFER_BIT); + } + } + C::ClearDepthAndStencil(depth, stencil_value) => { + // Prefer `clear` as `clear_buffer` functions have issues on Sandy Bridge + // on Windows. + unsafe { + gl.clear_depth_f32(depth); + gl.clear_stencil(stencil_value as i32); + gl.clear(glow::DEPTH_BUFFER_BIT | glow::STENCIL_BUFFER_BIT); + } + } + C::BufferBarrier(raw, usage) => { + let mut flags = 0; + if usage.contains(wgt::BufferUses::VERTEX) { + flags |= glow::VERTEX_ATTRIB_ARRAY_BARRIER_BIT; + unsafe { gl.bind_buffer(glow::ARRAY_BUFFER, Some(raw)) }; + unsafe { gl.vertex_attrib_pointer_f32(0, 1, glow::BYTE, true, 0, 0) }; + } + if usage.contains(wgt::BufferUses::INDEX) { + flags |= glow::ELEMENT_ARRAY_BARRIER_BIT; + unsafe { gl.bind_buffer(glow::ELEMENT_ARRAY_BUFFER, Some(raw)) }; + } + if usage.contains(wgt::BufferUses::UNIFORM) { + flags |= glow::UNIFORM_BARRIER_BIT; + } + if usage.contains(wgt::BufferUses::INDIRECT) { + flags |= glow::COMMAND_BARRIER_BIT; + unsafe { gl.bind_buffer(glow::DRAW_INDIRECT_BUFFER, Some(raw)) }; + } + if usage.contains(wgt::BufferUses::COPY_SRC) { + flags |= glow::PIXEL_BUFFER_BARRIER_BIT; + unsafe { gl.bind_buffer(glow::PIXEL_UNPACK_BUFFER, Some(raw)) }; + } + if usage.contains(wgt::BufferUses::COPY_DST) { + flags |= glow::PIXEL_BUFFER_BARRIER_BIT; + unsafe { gl.bind_buffer(glow::PIXEL_PACK_BUFFER, Some(raw)) }; + } + if usage.intersects(wgt::BufferUses::MAP_READ | wgt::BufferUses::MAP_WRITE) { + flags |= glow::BUFFER_UPDATE_BARRIER_BIT; + } + if usage.intersects( + wgt::BufferUses::STORAGE_READ_ONLY | wgt::BufferUses::STORAGE_READ_WRITE, + ) { + flags |= glow::SHADER_STORAGE_BARRIER_BIT; + } + unsafe { gl.memory_barrier(flags) }; + } + // because `STORAGE_WRITE_ONLY` and `STORAGE_READ_WRITE` are only states + // we can transit from due OpenGL memory barriers are used to make _subsequent_ + // operations see changes from the _shader_ side. We filter out usage changes that are + // does not comes from the shader side in `transition_textures` + C::TextureBarrier(usage) => { + let mut flags = 0; + if usage.contains(wgt::TextureUses::RESOURCE) { + flags |= glow::TEXTURE_FETCH_BARRIER_BIT; + } + if usage.intersects( + wgt::TextureUses::STORAGE_READ_ONLY + | wgt::TextureUses::STORAGE_WRITE_ONLY + | wgt::TextureUses::STORAGE_READ_WRITE, + ) { + flags |= glow::SHADER_IMAGE_ACCESS_BARRIER_BIT; + } + if usage.intersects(wgt::TextureUses::COPY_SRC) { + flags |= glow::PIXEL_BUFFER_BARRIER_BIT; + } + if usage.contains(wgt::TextureUses::COPY_DST) { + flags |= glow::TEXTURE_UPDATE_BARRIER_BIT; + } + if usage.intersects( + wgt::TextureUses::COLOR_TARGET + | wgt::TextureUses::DEPTH_STENCIL_READ + | wgt::TextureUses::DEPTH_STENCIL_WRITE, + ) { + flags |= glow::FRAMEBUFFER_BARRIER_BIT; + } + unsafe { gl.memory_barrier(flags) }; + } + C::SetViewport { + ref rect, + ref depth, + } => { + unsafe { gl.viewport(rect.x, rect.y, rect.w, rect.h) }; + unsafe { gl.depth_range_f32(depth.start, depth.end) }; + } + C::SetScissor(ref rect) => { + unsafe { gl.scissor(rect.x, rect.y, rect.w, rect.h) }; + unsafe { gl.enable(glow::SCISSOR_TEST) }; + } + C::SetStencilFunc { + face, + function, + reference, + read_mask, + } => { + unsafe { gl.stencil_func_separate(face, function, reference as i32, read_mask) }; + } + C::SetStencilOps { + face, + write_mask, + ref ops, + } => { + unsafe { gl.stencil_mask_separate(face, write_mask) }; + unsafe { gl.stencil_op_separate(face, ops.fail, ops.depth_fail, ops.pass) }; + } + C::SetVertexAttribute { + buffer, + ref buffer_desc, + attribute_desc: ref vat, + } => { + unsafe { gl.bind_buffer(glow::ARRAY_BUFFER, buffer) }; + unsafe { gl.enable_vertex_attrib_array(vat.location) }; + + if buffer.is_none() { + match vat.format_desc.attrib_kind { + super::VertexAttribKind::Float => unsafe { + gl.vertex_attrib_format_f32( + vat.location, + vat.format_desc.element_count, + vat.format_desc.element_format, + true, // always normalized + vat.offset, + ) + }, + super::VertexAttribKind::Integer => unsafe { + gl.vertex_attrib_format_i32( + vat.location, + vat.format_desc.element_count, + vat.format_desc.element_format, + vat.offset, + ) + }, + } + + //Note: there is apparently a bug on AMD 3500U: + // this call is ignored if the current array is disabled. + unsafe { gl.vertex_attrib_binding(vat.location, vat.buffer_index) }; + } else { + match vat.format_desc.attrib_kind { + super::VertexAttribKind::Float => unsafe { + gl.vertex_attrib_pointer_f32( + vat.location, + vat.format_desc.element_count, + vat.format_desc.element_format, + true, // always normalized + buffer_desc.stride as i32, + vat.offset as i32, + ) + }, + super::VertexAttribKind::Integer => unsafe { + gl.vertex_attrib_pointer_i32( + vat.location, + vat.format_desc.element_count, + vat.format_desc.element_format, + buffer_desc.stride as i32, + vat.offset as i32, + ) + }, + } + unsafe { gl.vertex_attrib_divisor(vat.location, buffer_desc.step as u32) }; + } + } + C::UnsetVertexAttribute(location) => { + unsafe { gl.disable_vertex_attrib_array(location) }; + } + C::SetVertexBuffer { + index, + ref buffer, + ref buffer_desc, + } => { + unsafe { gl.vertex_binding_divisor(index, buffer_desc.step as u32) }; + unsafe { + gl.bind_vertex_buffer( + index, + Some(buffer.raw), + buffer.offset as i32, + buffer_desc.stride as i32, + ) + }; + } + C::SetDepth(ref depth) => { + unsafe { gl.depth_func(depth.function) }; + unsafe { gl.depth_mask(depth.mask) }; + } + C::SetDepthBias(bias) => { + if bias.is_enabled() { + unsafe { gl.enable(glow::POLYGON_OFFSET_FILL) }; + unsafe { gl.polygon_offset(bias.slope_scale, bias.constant as f32) }; + } else { + unsafe { gl.disable(glow::POLYGON_OFFSET_FILL) }; + } + } + C::ConfigureDepthStencil(aspects) => { + if aspects.contains(crate::FormatAspects::DEPTH) { + unsafe { gl.enable(glow::DEPTH_TEST) }; + } else { + unsafe { gl.disable(glow::DEPTH_TEST) }; + } + if aspects.contains(crate::FormatAspects::STENCIL) { + unsafe { gl.enable(glow::STENCIL_TEST) }; + } else { + unsafe { gl.disable(glow::STENCIL_TEST) }; + } + } + C::SetAlphaToCoverage(enabled) => { + if enabled { + unsafe { gl.enable(glow::SAMPLE_ALPHA_TO_COVERAGE) }; + } else { + unsafe { gl.disable(glow::SAMPLE_ALPHA_TO_COVERAGE) }; + } + } + C::SetProgram(program) => { + unsafe { gl.use_program(Some(program)) }; + } + C::SetPrimitive(ref state) => { + unsafe { gl.front_face(state.front_face) }; + if state.cull_face != 0 { + unsafe { gl.enable(glow::CULL_FACE) }; + unsafe { gl.cull_face(state.cull_face) }; + } else { + unsafe { gl.disable(glow::CULL_FACE) }; + } + if self.features.contains(wgt::Features::DEPTH_CLIP_CONTROL) { + //Note: this is a bit tricky, since we are controlling the clip, not the clamp. + if state.unclipped_depth { + unsafe { gl.enable(glow::DEPTH_CLAMP) }; + } else { + unsafe { gl.disable(glow::DEPTH_CLAMP) }; + } + } + // POLYGON_MODE_LINE also implies POLYGON_MODE_POINT + if self.features.contains(wgt::Features::POLYGON_MODE_LINE) { + unsafe { gl.polygon_mode(glow::FRONT_AND_BACK, state.polygon_mode) }; + } + } + C::SetBlendConstant(c) => { + unsafe { gl.blend_color(c[0], c[1], c[2], c[3]) }; + } + C::SetColorTarget { + draw_buffer_index, + desc: super::ColorTargetDesc { mask, ref blend }, + } => { + use wgt::ColorWrites as Cw; + if let Some(index) = draw_buffer_index { + unsafe { + gl.color_mask_draw_buffer( + index, + mask.contains(Cw::RED), + mask.contains(Cw::GREEN), + mask.contains(Cw::BLUE), + mask.contains(Cw::ALPHA), + ) + }; + if let Some(ref blend) = *blend { + unsafe { gl.enable_draw_buffer(glow::BLEND, index) }; + if blend.color != blend.alpha { + unsafe { + gl.blend_equation_separate_draw_buffer( + index, + blend.color.equation, + blend.alpha.equation, + ) + }; + unsafe { + gl.blend_func_separate_draw_buffer( + index, + blend.color.src, + blend.color.dst, + blend.alpha.src, + blend.alpha.dst, + ) + }; + } else { + unsafe { gl.blend_equation_draw_buffer(index, blend.color.equation) }; + unsafe { + gl.blend_func_draw_buffer(index, blend.color.src, blend.color.dst) + }; + } + } else { + unsafe { gl.disable_draw_buffer(glow::BLEND, index) }; + } + } else { + unsafe { + gl.color_mask( + mask.contains(Cw::RED), + mask.contains(Cw::GREEN), + mask.contains(Cw::BLUE), + mask.contains(Cw::ALPHA), + ) + }; + if let Some(ref blend) = *blend { + unsafe { gl.enable(glow::BLEND) }; + if blend.color != blend.alpha { + unsafe { + gl.blend_equation_separate( + blend.color.equation, + blend.alpha.equation, + ) + }; + unsafe { + gl.blend_func_separate( + blend.color.src, + blend.color.dst, + blend.alpha.src, + blend.alpha.dst, + ) + }; + } else { + unsafe { gl.blend_equation(blend.color.equation) }; + unsafe { gl.blend_func(blend.color.src, blend.color.dst) }; + } + } else { + unsafe { gl.disable(glow::BLEND) }; + } + } + } + C::BindBuffer { + target, + slot, + buffer, + offset, + size, + } => { + unsafe { gl.bind_buffer_range(target, slot, Some(buffer), offset, size) }; + } + C::BindSampler(texture_index, sampler) => { + unsafe { gl.bind_sampler(texture_index, sampler) }; + } + C::BindTexture { + slot, + texture, + target, + aspects, + ref mip_levels, + } => { + unsafe { gl.active_texture(glow::TEXTURE0 + slot) }; + unsafe { gl.bind_texture(target, Some(texture)) }; + + unsafe { + gl.tex_parameter_i32(target, glow::TEXTURE_BASE_LEVEL, mip_levels.start as i32) + }; + unsafe { + gl.tex_parameter_i32( + target, + glow::TEXTURE_MAX_LEVEL, + (mip_levels.end - 1) as i32, + ) + }; + + let version = gl.version(); + let is_min_es_3_1 = version.is_embedded && (version.major, version.minor) >= (3, 1); + let is_min_4_3 = !version.is_embedded && (version.major, version.minor) >= (4, 3); + if is_min_es_3_1 || is_min_4_3 { + let mode = match aspects { + crate::FormatAspects::DEPTH => Some(glow::DEPTH_COMPONENT), + crate::FormatAspects::STENCIL => Some(glow::STENCIL_INDEX), + _ => None, + }; + if let Some(mode) = mode { + unsafe { + gl.tex_parameter_i32( + target, + glow::DEPTH_STENCIL_TEXTURE_MODE, + mode as _, + ) + }; + } + } + } + C::BindImage { slot, ref binding } => { + unsafe { + gl.bind_image_texture( + slot, + Some(binding.raw), + binding.mip_level as i32, + binding.array_layer.is_none(), + binding.array_layer.unwrap_or_default() as i32, + binding.access, + binding.format, + ) + }; + } + C::InsertDebugMarker(ref range) => { + let marker = extract_marker(data_bytes, range); + unsafe { + if self + .shared + .private_caps + .contains(PrivateCapabilities::DEBUG_FNS) + { + gl.debug_message_insert( + glow::DEBUG_SOURCE_APPLICATION, + glow::DEBUG_TYPE_MARKER, + DEBUG_ID, + glow::DEBUG_SEVERITY_NOTIFICATION, + to_debug_str(marker), + ) + } + }; + } + C::PushDebugGroup(ref range) => { + let marker = extract_marker(data_bytes, range); + unsafe { + if self + .shared + .private_caps + .contains(PrivateCapabilities::DEBUG_FNS) + { + gl.push_debug_group( + glow::DEBUG_SOURCE_APPLICATION, + DEBUG_ID, + to_debug_str(marker), + ) + } + }; + } + C::PopDebugGroup => { + unsafe { + if self + .shared + .private_caps + .contains(PrivateCapabilities::DEBUG_FNS) + { + gl.pop_debug_group() + } + }; + } + C::SetImmediates { + ref uniform, + offset, + } => { + fn get_data(data: &[u8], offset: u32) -> [T; COUNT] + where + [T; COUNT]: bytemuck::AnyBitPattern, + { + let data_required = size_of::() * COUNT; + let raw = &data[(offset as usize)..][..data_required]; + bytemuck::pod_read_unaligned(raw) + } + + let location = Some(&uniform.location); + + match uniform.ty { + // + // --- Float 1-4 Component --- + // + naga::TypeInner::Scalar(naga::Scalar::F32) => { + let data = get_data::(data_bytes, offset)[0]; + unsafe { gl.uniform_1_f32(location, data) }; + } + naga::TypeInner::Vector { + size: naga::VectorSize::Bi, + scalar: naga::Scalar::F32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_2_f32_slice(location, data) }; + } + naga::TypeInner::Vector { + size: naga::VectorSize::Tri, + scalar: naga::Scalar::F32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_3_f32_slice(location, data) }; + } + naga::TypeInner::Vector { + size: naga::VectorSize::Quad, + scalar: naga::Scalar::F32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_4_f32_slice(location, data) }; + } + + // + // --- Int 1-4 Component --- + // + naga::TypeInner::Scalar(naga::Scalar::I32) => { + let data = get_data::(data_bytes, offset)[0]; + unsafe { gl.uniform_1_i32(location, data) }; + } + naga::TypeInner::Vector { + size: naga::VectorSize::Bi, + scalar: naga::Scalar::I32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_2_i32_slice(location, data) }; + } + naga::TypeInner::Vector { + size: naga::VectorSize::Tri, + scalar: naga::Scalar::I32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_3_i32_slice(location, data) }; + } + naga::TypeInner::Vector { + size: naga::VectorSize::Quad, + scalar: naga::Scalar::I32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_4_i32_slice(location, data) }; + } + + // + // --- Uint 1-4 Component --- + // + naga::TypeInner::Scalar(naga::Scalar::U32) => { + let data = get_data::(data_bytes, offset)[0]; + unsafe { gl.uniform_1_u32(location, data) }; + } + naga::TypeInner::Vector { + size: naga::VectorSize::Bi, + scalar: naga::Scalar::U32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_2_u32_slice(location, data) }; + } + naga::TypeInner::Vector { + size: naga::VectorSize::Tri, + scalar: naga::Scalar::U32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_3_u32_slice(location, data) }; + } + naga::TypeInner::Vector { + size: naga::VectorSize::Quad, + scalar: naga::Scalar::U32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_4_u32_slice(location, data) }; + } + + // + // --- Matrix 2xR --- + // + naga::TypeInner::Matrix { + columns: naga::VectorSize::Bi, + rows: naga::VectorSize::Bi, + scalar: naga::Scalar::F32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_matrix_2_f32_slice(location, false, data) }; + } + naga::TypeInner::Matrix { + columns: naga::VectorSize::Bi, + rows: naga::VectorSize::Tri, + scalar: naga::Scalar::F32, + } => { + // repack 2 vec3s into 6 values. + let unpacked_data = &get_data::(data_bytes, offset); + #[rustfmt::skip] + let packed_data = [ + unpacked_data[0], unpacked_data[1], unpacked_data[2], + unpacked_data[4], unpacked_data[5], unpacked_data[6], + ]; + unsafe { gl.uniform_matrix_2x3_f32_slice(location, false, &packed_data) }; + } + naga::TypeInner::Matrix { + columns: naga::VectorSize::Bi, + rows: naga::VectorSize::Quad, + scalar: naga::Scalar::F32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_matrix_2x4_f32_slice(location, false, data) }; + } + + // + // --- Matrix 3xR --- + // + naga::TypeInner::Matrix { + columns: naga::VectorSize::Tri, + rows: naga::VectorSize::Bi, + scalar: naga::Scalar::F32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_matrix_3x2_f32_slice(location, false, data) }; + } + naga::TypeInner::Matrix { + columns: naga::VectorSize::Tri, + rows: naga::VectorSize::Tri, + scalar: naga::Scalar::F32, + } => { + // repack 3 vec3s into 9 values. + let unpacked_data = &get_data::(data_bytes, offset); + #[rustfmt::skip] + let packed_data = [ + unpacked_data[0], unpacked_data[1], unpacked_data[2], + unpacked_data[4], unpacked_data[5], unpacked_data[6], + unpacked_data[8], unpacked_data[9], unpacked_data[10], + ]; + unsafe { gl.uniform_matrix_3_f32_slice(location, false, &packed_data) }; + } + naga::TypeInner::Matrix { + columns: naga::VectorSize::Tri, + rows: naga::VectorSize::Quad, + scalar: naga::Scalar::F32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_matrix_3x4_f32_slice(location, false, data) }; + } + + // + // --- Matrix 4xR --- + // + naga::TypeInner::Matrix { + columns: naga::VectorSize::Quad, + rows: naga::VectorSize::Bi, + scalar: naga::Scalar::F32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_matrix_4x2_f32_slice(location, false, data) }; + } + naga::TypeInner::Matrix { + columns: naga::VectorSize::Quad, + rows: naga::VectorSize::Tri, + scalar: naga::Scalar::F32, + } => { + // repack 4 vec3s into 12 values. + let unpacked_data = &get_data::(data_bytes, offset); + #[rustfmt::skip] + let packed_data = [ + unpacked_data[0], unpacked_data[1], unpacked_data[2], + unpacked_data[4], unpacked_data[5], unpacked_data[6], + unpacked_data[8], unpacked_data[9], unpacked_data[10], + unpacked_data[12], unpacked_data[13], unpacked_data[14], + ]; + unsafe { gl.uniform_matrix_4x3_f32_slice(location, false, &packed_data) }; + } + naga::TypeInner::Matrix { + columns: naga::VectorSize::Quad, + rows: naga::VectorSize::Quad, + scalar: naga::Scalar::F32, + } => { + let data = &get_data::(data_bytes, offset); + unsafe { gl.uniform_matrix_4_f32_slice(location, false, data) }; + } + _ => panic!("Unsupported uniform datatype: {:?}!", uniform.ty), + } + } + C::SetClipDistances { + old_count, + new_count, + } => { + // Disable clip planes that are no longer active + for i in new_count..old_count { + unsafe { gl.disable(glow::CLIP_DISTANCE0 + i) }; + } + + // Enable clip planes that are now active + for i in old_count..new_count { + unsafe { gl.enable(glow::CLIP_DISTANCE0 + i) }; + } + } + } + } +} + +impl crate::Queue for super::Queue { + type A = super::Api; + + unsafe fn submit( + &self, + command_buffers: &[&super::CommandBuffer], + _surface_textures: &[&super::Texture], + (signal_fence, signal_value): (&mut super::Fence, crate::FenceValue), + ) -> Result<(), crate::DeviceError> { + let shared = Arc::clone(&self.shared); + let gl = &shared.context.lock(); + for cmd_buf in command_buffers.iter() { + // The command encoder assumes a default state when encoding the command buffer. + // Always reset the state between command_buffers to reflect this assumption. Do + // this at the beginning of the loop in case something outside of wgpu modified + // this state prior to commit. + unsafe { self.reset_state(gl) }; + if let Some(ref label) = cmd_buf.label { + if self + .shared + .private_caps + .contains(PrivateCapabilities::DEBUG_FNS) + { + unsafe { + gl.push_debug_group( + glow::DEBUG_SOURCE_APPLICATION, + DEBUG_ID, + to_debug_str(label), + ) + }; + } + } + + for command in cmd_buf.commands.iter() { + unsafe { self.process(gl, command, &cmd_buf.data_bytes, &cmd_buf.queries) }; + } + + if cmd_buf.label.is_some() + && self + .shared + .private_caps + .contains(PrivateCapabilities::DEBUG_FNS) + { + unsafe { gl.pop_debug_group() }; + } + } + + signal_fence.maintain(gl); + signal_fence.signal(gl, signal_value)?; + + // This is extremely important. If we don't flush, the above fences may never + // be signaled, particularly in headless contexts. Headed contexts will + // often flush every so often, but headless contexts may not. + unsafe { gl.flush() }; + + Ok(()) + } + + unsafe fn present( + &self, + surface: &super::Surface, + texture: super::Texture, + ) -> Result<(), crate::SurfaceError> { + unsafe { surface.present(texture, &self.shared.context) } + } + + unsafe fn get_timestamp_period(&self) -> f32 { + 1.0 + } +} + +#[cfg(send_sync)] +unsafe impl Sync for super::Queue {} +#[cfg(send_sync)] +unsafe impl Send for super::Queue {} diff --git a/third_party/wgpu-hal-29.0.4/src/gles/shaders/clear.frag b/third_party/wgpu-hal-29.0.4/src/gles/shaders/clear.frag new file mode 100644 index 0000000..1d0e414 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/shaders/clear.frag @@ -0,0 +1,7 @@ +uniform vec4 color; +//Hack: Some WebGL implementations don't find "color" otherwise. +uniform vec4 color_workaround; +out vec4 frag; +void main() { + frag = color + color_workaround; +} diff --git a/third_party/wgpu-hal-29.0.4/src/gles/shaders/clear.vert b/third_party/wgpu-hal-29.0.4/src/gles/shaders/clear.vert new file mode 100644 index 0000000..341b4e5 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/shaders/clear.vert @@ -0,0 +1,9 @@ +// A triangle that fills the whole screen +vec2[3] TRIANGLE_POS = vec2[]( + vec2( 0.0, -3.0), + vec2(-3.0, 1.0), + vec2( 3.0, 1.0) +); +void main() { + gl_Position = vec4(TRIANGLE_POS[gl_VertexID], 0.0, 1.0); +} \ No newline at end of file diff --git a/third_party/wgpu-hal-29.0.4/src/gles/shaders/srgb_present.frag b/third_party/wgpu-hal-29.0.4/src/gles/shaders/srgb_present.frag new file mode 100644 index 0000000..853f82a --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/shaders/srgb_present.frag @@ -0,0 +1,16 @@ +#version 300 es +precision mediump float; +in vec2 uv; +uniform sampler2D present_texture; +out vec4 frag; +vec4 linear_to_srgb(vec4 linear) { + vec3 color_linear = linear.rgb; + vec3 selector = ceil(color_linear - 0.0031308); // 0 if under value, 1 if over + vec3 under = 12.92 * color_linear; + vec3 over = 1.055 * pow(color_linear, vec3(0.41666)) - 0.055; + vec3 result = mix(under, over, selector); + return vec4(result, linear.a); +} +void main() { + frag = linear_to_srgb(texture(present_texture, uv)); +} \ No newline at end of file diff --git a/third_party/wgpu-hal-29.0.4/src/gles/shaders/srgb_present.vert b/third_party/wgpu-hal-29.0.4/src/gles/shaders/srgb_present.vert new file mode 100644 index 0000000..922f2a1 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/shaders/srgb_present.vert @@ -0,0 +1,18 @@ +#version 300 es +precision mediump float; +// A triangle that fills the whole screen +const vec2[3] TRIANGLE_POS = vec2[]( + vec2( 0.0, -3.0), + vec2(-3.0, 1.0), + vec2( 3.0, 1.0) +); +const vec2[3] TRIANGLE_UV = vec2[]( + vec2( 0.5, 1.), + vec2( -1.0, -1.0), + vec2( 2.0, -1.0) +); +out vec2 uv; +void main() { + uv = TRIANGLE_UV[gl_VertexID]; + gl_Position = vec4(TRIANGLE_POS[gl_VertexID], 0.0, 1.0); +} \ No newline at end of file diff --git a/third_party/wgpu-hal-29.0.4/src/gles/web.rs b/third_party/wgpu-hal-29.0.4/src/gles/web.rs new file mode 100644 index 0000000..8e404d6 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/web.rs @@ -0,0 +1,453 @@ +use alloc::{format, string::String, vec::Vec}; + +use glow::HasContext; +use parking_lot::{Mutex, RwLock}; +use wasm_bindgen::{JsCast, JsValue}; + +use super::TextureFormatDesc; + +/// A wrapper around a [`glow::Context`] to provide a fake `lock()` api that makes it compatible +/// with the `AdapterContext` API from the EGL implementation. +pub struct AdapterContext { + pub glow_context: glow::Context, + pub webgl2_context: web_sys::WebGl2RenderingContext, +} + +impl AdapterContext { + pub fn is_owned(&self) -> bool { + false + } + + /// Obtain a lock to the EGL context and get handle to the [`glow::Context`] that can be used to + /// do rendering. + #[track_caller] + pub fn lock(&self) -> &glow::Context { + &self.glow_context + } +} + +#[derive(Debug)] +pub struct Instance { + options: wgt::GlBackendOptions, +} + +impl Instance { + pub fn create_surface_from_canvas( + &self, + canvas: web_sys::HtmlCanvasElement, + ) -> Result { + let result = + canvas.get_context_with_context_options("webgl2", &Self::create_context_options()); + self.create_surface_from_context(Canvas::Canvas(canvas), result) + } + + pub fn create_surface_from_offscreen_canvas( + &self, + canvas: web_sys::OffscreenCanvas, + ) -> Result { + let result = + canvas.get_context_with_context_options("webgl2", &Self::create_context_options()); + self.create_surface_from_context(Canvas::Offscreen(canvas), result) + } + + /// Common portion of public `create_surface_from_*` functions. + /// + /// Note: Analogous code also exists in the WebGPU backend at + /// `wgpu::backend::web::Context`. + fn create_surface_from_context( + &self, + canvas: Canvas, + context_result: Result, JsValue>, + ) -> Result { + let context_object: js_sys::Object = match context_result { + Ok(Some(context)) => context, + Ok(None) => { + // + // A getContext() call “returns null if contextId is not supported, or if the + // canvas has already been initialized with another context type”. Additionally, + // “not supported” could include “insufficient GPU resources” or “the GPU process + // previously crashed”. So, we must return it as an `Err` since it could occur + // for circumstances outside the application author's control. + return Err(crate::InstanceError::new(String::from(concat!( + "canvas.getContext() returned null; ", + "webgl2 not available or canvas already in use" + )))); + } + Err(js_error) => { + // + // A thrown exception indicates misuse of the canvas state. + return Err(crate::InstanceError::new(format!( + "canvas.getContext() threw exception {js_error:?}", + ))); + } + }; + + // Not returning this error because it is a type error that shouldn't happen unless + // the browser, JS builtin objects, or wasm bindings are misbehaving somehow. + let webgl2_context: web_sys::WebGl2RenderingContext = context_object + .dyn_into() + .expect("canvas context is not a WebGl2RenderingContext"); + + Ok(Surface { + canvas, + webgl2_context, + srgb_present_program: Mutex::new(None), + swapchain: RwLock::new(None), + texture: Mutex::new(None), + presentable: true, + }) + } + + fn create_context_options() -> js_sys::Object { + let context_options = js_sys::Object::new(); + js_sys::Reflect::set(&context_options, &"antialias".into(), &JsValue::FALSE) + .expect("Cannot create context options"); + context_options + } +} + +#[cfg(send_sync)] +unsafe impl Sync for Instance {} +#[cfg(send_sync)] +unsafe impl Send for Instance {} + +impl crate::Instance for Instance { + type A = super::Api; + + unsafe fn init(desc: &crate::InstanceDescriptor<'_>) -> Result { + profiling::scope!("Init OpenGL (WebGL) Backend"); + Ok(Instance { + options: desc.backend_options.gl.clone(), + }) + } + + unsafe fn enumerate_adapters( + &self, + surface_hint: Option<&Surface>, + ) -> Vec> { + if let Some(surface_hint) = surface_hint { + let gl = glow::Context::from_webgl2_context(surface_hint.webgl2_context.clone()); + + unsafe { + super::Adapter::expose( + AdapterContext { + glow_context: gl, + webgl2_context: surface_hint.webgl2_context.clone(), + }, + self.options.clone(), + ) + } + .into_iter() + .collect() + } else { + Vec::new() + } + } + + unsafe fn create_surface( + &self, + _display_handle: raw_window_handle::RawDisplayHandle, + window_handle: raw_window_handle::RawWindowHandle, + ) -> Result { + let canvas: web_sys::HtmlCanvasElement = match window_handle { + raw_window_handle::RawWindowHandle::Web(handle) => web_sys::window() + .and_then(|win| win.document()) + .expect("Cannot get document") + .query_selector(&format!("canvas[data-raw-handle=\"{}\"]", handle.id)) + .expect("Cannot query for canvas") + .expect("Canvas is not found") + .dyn_into() + .expect("Failed to downcast to canvas type"), + raw_window_handle::RawWindowHandle::WebCanvas(handle) => { + let value: &JsValue = unsafe { handle.obj.cast().as_ref() }; + value.clone().unchecked_into() + } + raw_window_handle::RawWindowHandle::WebOffscreenCanvas(handle) => { + let value: &JsValue = unsafe { handle.obj.cast().as_ref() }; + let canvas: web_sys::OffscreenCanvas = value.clone().unchecked_into(); + + return self.create_surface_from_offscreen_canvas(canvas); + } + _ => { + return Err(crate::InstanceError::new(format!( + "window handle {window_handle:?} is not a web handle" + ))) + } + }; + + self.create_surface_from_canvas(canvas) + } +} + +#[derive(Debug)] +pub struct Surface { + canvas: Canvas, + pub(super) webgl2_context: web_sys::WebGl2RenderingContext, + pub(super) swapchain: RwLock>, + texture: Mutex>, + pub(super) presentable: bool, + srgb_present_program: Mutex>, +} + +impl Clone for Surface { + fn clone(&self) -> Self { + Self { + canvas: self.canvas.clone(), + webgl2_context: self.webgl2_context.clone(), + swapchain: RwLock::new(self.swapchain.read().clone()), + texture: Mutex::new(*self.texture.lock()), + presentable: self.presentable, + srgb_present_program: Mutex::new(*self.srgb_present_program.lock()), + } + } +} + +#[cfg(send_sync)] +unsafe impl Sync for Surface {} +#[cfg(send_sync)] +unsafe impl Send for Surface {} + +#[derive(Clone, Debug)] +enum Canvas { + Canvas(web_sys::HtmlCanvasElement), + Offscreen(web_sys::OffscreenCanvas), +} + +#[derive(Clone, Debug)] +pub struct Swapchain { + pub(crate) extent: wgt::Extent3d, + // pub(crate) channel: f::ChannelType, + pub(super) format: wgt::TextureFormat, + pub(super) framebuffer: glow::Framebuffer, + pub(super) format_desc: TextureFormatDesc, +} + +impl Surface { + pub(super) unsafe fn present( + &self, + _suf_texture: super::Texture, + context: &AdapterContext, + ) -> Result<(), crate::SurfaceError> { + let gl = &context.glow_context; + let swapchain = self.swapchain.read(); + let swapchain = swapchain.as_ref().ok_or(crate::SurfaceError::Other( + "need to configure surface before presenting", + ))?; + + if swapchain.format.is_srgb() { + // Important to set the viewport since we don't know in what state the user left it. + unsafe { + gl.viewport( + 0, + 0, + swapchain.extent.width as _, + swapchain.extent.height as _, + ) + }; + unsafe { gl.bind_framebuffer(glow::DRAW_FRAMEBUFFER, None) }; + unsafe { gl.bind_sampler(0, None) }; + unsafe { gl.active_texture(glow::TEXTURE0) }; + unsafe { gl.bind_texture(glow::TEXTURE_2D, *self.texture.lock()) }; + unsafe { gl.use_program(*self.srgb_present_program.lock()) }; + unsafe { gl.disable(glow::DEPTH_TEST) }; + unsafe { gl.disable(glow::STENCIL_TEST) }; + unsafe { gl.disable(glow::SCISSOR_TEST) }; + unsafe { gl.disable(glow::BLEND) }; + unsafe { gl.disable(glow::CULL_FACE) }; + unsafe { gl.draw_buffers(&[glow::BACK]) }; + unsafe { gl.draw_arrays(glow::TRIANGLES, 0, 3) }; + } else { + unsafe { gl.bind_framebuffer(glow::READ_FRAMEBUFFER, Some(swapchain.framebuffer)) }; + unsafe { gl.bind_framebuffer(glow::DRAW_FRAMEBUFFER, None) }; + // Note the Y-flipping here. GL's presentation is not flipped, + // but main rendering is. Therefore, we Y-flip the output positions + // in the shader, and also this blit. + unsafe { + gl.blit_framebuffer( + 0, + swapchain.extent.height as i32, + swapchain.extent.width as i32, + 0, + 0, + 0, + swapchain.extent.width as i32, + swapchain.extent.height as i32, + glow::COLOR_BUFFER_BIT, + glow::NEAREST, + ) + }; + } + + Ok(()) + } + + unsafe fn create_srgb_present_program(gl: &glow::Context) -> glow::Program { + let program = unsafe { gl.create_program() }.expect("Could not create shader program"); + let vertex = + unsafe { gl.create_shader(glow::VERTEX_SHADER) }.expect("Could not create shader"); + unsafe { gl.shader_source(vertex, include_str!("./shaders/srgb_present.vert")) }; + unsafe { gl.compile_shader(vertex) }; + let fragment = + unsafe { gl.create_shader(glow::FRAGMENT_SHADER) }.expect("Could not create shader"); + unsafe { gl.shader_source(fragment, include_str!("./shaders/srgb_present.frag")) }; + unsafe { gl.compile_shader(fragment) }; + unsafe { gl.attach_shader(program, vertex) }; + unsafe { gl.attach_shader(program, fragment) }; + unsafe { gl.link_program(program) }; + unsafe { gl.delete_shader(vertex) }; + unsafe { gl.delete_shader(fragment) }; + unsafe { gl.bind_texture(glow::TEXTURE_2D, None) }; + + program + } + + pub fn supports_srgb(&self) -> bool { + // present.frag takes care of handling srgb conversion + true + } +} + +impl crate::Surface for Surface { + type A = super::Api; + + unsafe fn configure( + &self, + device: &super::Device, + config: &crate::SurfaceConfiguration, + ) -> Result<(), crate::SurfaceError> { + match self.canvas { + Canvas::Canvas(ref canvas) => { + canvas.set_width(config.extent.width); + canvas.set_height(config.extent.height); + } + Canvas::Offscreen(ref canvas) => { + canvas.set_width(config.extent.width); + canvas.set_height(config.extent.height); + } + } + + let gl = &device.shared.context.lock(); + + { + let mut swapchain = self.swapchain.write(); + if let Some(swapchain) = swapchain.take() { + // delete all frame buffers already allocated + unsafe { gl.delete_framebuffer(swapchain.framebuffer) }; + } + } + { + let mut srgb_present_program = self.srgb_present_program.lock(); + if srgb_present_program.is_none() && config.format.is_srgb() { + *srgb_present_program = Some(unsafe { Self::create_srgb_present_program(gl) }); + } + } + { + let mut texture = self.texture.lock(); + if let Some(texture) = texture.take() { + unsafe { gl.delete_texture(texture) }; + } + + *texture = Some(unsafe { gl.create_texture() }.map_err(|error| { + log::error!("Internal swapchain texture creation failed: {error}"); + crate::DeviceError::OutOfMemory + })?); + + let desc = device.shared.describe_texture_format(config.format); + unsafe { gl.bind_texture(glow::TEXTURE_2D, *texture) }; + unsafe { + gl.tex_parameter_i32( + glow::TEXTURE_2D, + glow::TEXTURE_MIN_FILTER, + glow::NEAREST as _, + ) + }; + unsafe { + gl.tex_parameter_i32( + glow::TEXTURE_2D, + glow::TEXTURE_MAG_FILTER, + glow::NEAREST as _, + ) + }; + unsafe { + gl.tex_storage_2d( + glow::TEXTURE_2D, + 1, + desc.internal, + config.extent.width as i32, + config.extent.height as i32, + ) + }; + + let framebuffer = unsafe { gl.create_framebuffer() }.map_err(|error| { + log::error!("Internal swapchain framebuffer creation failed: {error}"); + crate::DeviceError::OutOfMemory + })?; + unsafe { gl.bind_framebuffer(glow::READ_FRAMEBUFFER, Some(framebuffer)) }; + unsafe { + gl.framebuffer_texture_2d( + glow::READ_FRAMEBUFFER, + glow::COLOR_ATTACHMENT0, + glow::TEXTURE_2D, + *texture, + 0, + ) + }; + unsafe { gl.bind_texture(glow::TEXTURE_2D, None) }; + + let mut swapchain = self.swapchain.write(); + *swapchain = Some(Swapchain { + extent: config.extent, + // channel: config.format.base_format().1, + format: config.format, + format_desc: desc, + framebuffer, + }); + } + + Ok(()) + } + + unsafe fn unconfigure(&self, device: &super::Device) { + let gl = device.shared.context.lock(); + { + let mut swapchain = self.swapchain.write(); + if let Some(swapchain) = swapchain.take() { + unsafe { gl.delete_framebuffer(swapchain.framebuffer) }; + } + } + if let Some(renderbuffer) = self.texture.lock().take() { + unsafe { gl.delete_texture(renderbuffer) }; + } + } + + unsafe fn acquire_texture( + &self, + _timeout_ms: Option, //TODO + _fence: &super::Fence, + ) -> Result, crate::SurfaceError> { + let swapchain = self.swapchain.read(); + let sc = swapchain.as_ref().unwrap(); + let texture = super::Texture { + inner: super::TextureInner::Texture { + raw: self.texture.lock().unwrap(), + target: glow::TEXTURE_2D, + }, + drop_guard: None, + array_layer_count: 1, + mip_level_count: 1, + format: sc.format, + format_desc: sc.format_desc.clone(), + copy_size: crate::CopyExtent { + width: sc.extent.width, + height: sc.extent.height, + depth: 1, + }, + }; + Ok(crate::AcquiredSurfaceTexture { + texture, + suboptimal: false, + }) + } + + unsafe fn discard_texture(&self, _texture: super::Texture) {} +} diff --git a/third_party/wgpu-hal-29.0.4/src/gles/wgl.rs b/third_party/wgpu-hal-29.0.4/src/gles/wgl.rs new file mode 100644 index 0000000..dba309e --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/gles/wgl.rs @@ -0,0 +1,900 @@ +use alloc::{borrow::ToOwned as _, ffi::CString, string::String, sync::Arc, vec::Vec}; +use core::{ + ffi::{c_int, c_void, CStr}, + mem::{self, ManuallyDrop}, + ptr, + time::Duration, +}; +use std::{ + sync::{ + mpsc::{sync_channel, SyncSender}, + LazyLock, + }, + thread, +}; + +use glow::HasContext; +use glutin_wgl_sys::wgl_extra::{ + Wgl, CONTEXT_CORE_PROFILE_BIT_ARB, CONTEXT_DEBUG_BIT_ARB, CONTEXT_FLAGS_ARB, + CONTEXT_PROFILE_MASK_ARB, +}; +use hashbrown::HashSet; +use parking_lot::{Mutex, MutexGuard, RwLock}; +use raw_window_handle::{RawDisplayHandle, RawWindowHandle}; +use wgt::InstanceFlags; +use windows::{ + core::{Error, PCSTR}, + Win32::{ + Foundation, + Graphics::{Gdi, OpenGL}, + System::LibraryLoader, + UI::WindowsAndMessaging, + }, +}; + +/// The amount of time to wait while trying to obtain a lock to the adapter context +const CONTEXT_LOCK_TIMEOUT_SECS: u64 = 1; + +/// A wrapper around a `[`glow::Context`]` and the required WGL context that uses locking to +/// guarantee exclusive access when shared with multiple threads. +pub struct AdapterContext { + inner: Arc>, +} + +unsafe impl Sync for AdapterContext {} +unsafe impl Send for AdapterContext {} + +impl AdapterContext { + pub fn is_owned(&self) -> bool { + true + } + + pub fn raw_context(&self) -> *mut c_void { + match self.inner.lock().context { + Some(ref wgl) => wgl.context.0, + None => ptr::null_mut(), + } + } + + /// Obtain a lock to the WGL context and get handle to the [`glow::Context`] that can be used to + /// do rendering. + #[track_caller] + pub fn lock(&self) -> AdapterContextLock<'_> { + let inner = self + .inner + // Don't lock forever. If it takes longer than 1 second to get the lock we've got a + // deadlock and should panic to show where we got stuck + .try_lock_for(Duration::from_secs(CONTEXT_LOCK_TIMEOUT_SECS)) + .expect("Could not lock adapter context. This is most-likely a deadlock."); + + if let Some(wgl) = &inner.context { + wgl.make_current(inner.device.dc).unwrap() + }; + + AdapterContextLock { inner } + } + + /// Obtain a lock to the WGL context and get handle to the [`glow::Context`] that can be used to + /// do rendering. + /// + /// Unlike [`lock`](Self::lock), this accepts a device to pass to `make_current` and exposes the error + /// when `make_current` fails. + #[track_caller] + fn lock_with_dc(&self, device: Gdi::HDC) -> windows::core::Result> { + let inner = self + .inner + .try_lock_for(Duration::from_secs(CONTEXT_LOCK_TIMEOUT_SECS)) + .expect("Could not lock adapter context. This is most-likely a deadlock."); + + if let Some(wgl) = &inner.context { + wgl.make_current(device)?; + } + + Ok(AdapterContextLock { inner }) + } +} + +/// A guard containing a lock to an [`AdapterContext`], while the GL context is kept current. +pub struct AdapterContextLock<'a> { + inner: MutexGuard<'a, Inner>, +} + +impl<'a> core::ops::Deref for AdapterContextLock<'a> { + type Target = glow::Context; + + fn deref(&self) -> &Self::Target { + &self.inner.gl + } +} + +impl<'a> Drop for AdapterContextLock<'a> { + fn drop(&mut self) { + if let Some(wgl) = &self.inner.context { + wgl.unmake_current().unwrap() + } + } +} + +struct WglContext { + context: OpenGL::HGLRC, +} + +impl WglContext { + fn make_current(&self, device: Gdi::HDC) -> windows::core::Result<()> { + unsafe { OpenGL::wglMakeCurrent(device, self.context) } + } + + fn unmake_current(&self) -> windows::core::Result<()> { + if unsafe { OpenGL::wglGetCurrentContext() }.is_invalid() { + return Ok(()); + } + unsafe { OpenGL::wglMakeCurrent(Default::default(), Default::default()) } + } +} + +impl Drop for WglContext { + fn drop(&mut self) { + if let Err(e) = unsafe { OpenGL::wglDeleteContext(self.context) } { + log::error!("failed to delete WGL context: {e}"); + } + } +} + +unsafe impl Send for WglContext {} +unsafe impl Sync for WglContext {} + +struct Inner { + gl: ManuallyDrop, + device: InstanceDevice, + context: Option, +} + +impl Drop for Inner { + fn drop(&mut self) { + struct CurrentGuard<'a>(&'a WglContext); + impl Drop for CurrentGuard<'_> { + fn drop(&mut self) { + self.0.unmake_current().unwrap(); + } + } + + // Context must be current when dropped. See safety docs on + // `glow::HasContext`. + // + // NOTE: This is only set to `None` by `Adapter::new_external` which + // requires the context to be current when anything that may be holding + // the `Arc` is dropped. + let _guard = self.context.as_ref().map(|wgl| { + wgl.make_current(self.device.dc).unwrap(); + CurrentGuard(wgl) + }); + // SAFETY: Field not used after this. + unsafe { ManuallyDrop::drop(&mut self.gl) }; + } +} + +unsafe impl Send for Inner {} +unsafe impl Sync for Inner {} + +pub struct Instance { + srgb_capable: bool, + options: wgt::GlBackendOptions, + inner: Arc>, +} + +unsafe impl Send for Instance {} +unsafe impl Sync for Instance {} + +fn load_gl_func(name: &str, module: Option) -> *const c_void { + let addr = CString::new(name.as_bytes()).unwrap(); + let mut ptr = unsafe { OpenGL::wglGetProcAddress(PCSTR(addr.as_ptr().cast())) }; + if ptr.is_none() { + if let Some(module) = module { + ptr = unsafe { LibraryLoader::GetProcAddress(module, PCSTR(addr.as_ptr().cast())) }; + } + } + ptr.map_or_else(ptr::null_mut, |p| p as *mut c_void) +} + +fn get_extensions(extra: &Wgl, dc: Gdi::HDC) -> HashSet { + if extra.GetExtensionsStringARB.is_loaded() { + unsafe { CStr::from_ptr(extra.GetExtensionsStringARB(dc.0)) } + .to_str() + .unwrap_or("") + } else { + "" + } + .split(' ') + .map(|s| s.to_owned()) + .collect() +} + +unsafe fn setup_pixel_format(dc: Gdi::HDC) -> Result<(), crate::InstanceError> { + { + let format = OpenGL::PIXELFORMATDESCRIPTOR { + nVersion: 1, + nSize: size_of::() as u16, + dwFlags: OpenGL::PFD_DRAW_TO_WINDOW + | OpenGL::PFD_SUPPORT_OPENGL + | OpenGL::PFD_DOUBLEBUFFER, + iPixelType: OpenGL::PFD_TYPE_RGBA, + cColorBits: 8, + ..unsafe { mem::zeroed() } + }; + + let index = unsafe { OpenGL::ChoosePixelFormat(dc, &format) }; + if index == 0 { + return Err(crate::InstanceError::with_source( + String::from("unable to choose pixel format"), + Error::from_thread(), + )); + } + + let current = unsafe { OpenGL::GetPixelFormat(dc) }; + + if index != current { + unsafe { OpenGL::SetPixelFormat(dc, index, &format) }.map_err(|e| { + crate::InstanceError::with_source(String::from("unable to set pixel format"), e) + })?; + } + } + + { + let index = unsafe { OpenGL::GetPixelFormat(dc) }; + if index == 0 { + return Err(crate::InstanceError::with_source( + String::from("unable to get pixel format index"), + Error::from_thread(), + )); + } + let mut format = Default::default(); + if unsafe { + OpenGL::DescribePixelFormat(dc, index, size_of_val(&format) as u32, Some(&mut format)) + } == 0 + { + return Err(crate::InstanceError::with_source( + String::from("unable to read pixel format"), + Error::from_thread(), + )); + } + + if !format.dwFlags.contains(OpenGL::PFD_SUPPORT_OPENGL) + || format.iPixelType != OpenGL::PFD_TYPE_RGBA + { + return Err(crate::InstanceError::new(String::from( + "unsuitable pixel format", + ))); + } + } + Ok(()) +} + +fn create_global_window_class() -> Result { + let instance = unsafe { LibraryLoader::GetModuleHandleA(None) }.map_err(|e| { + crate::InstanceError::with_source(String::from("unable to get executable instance"), e) + })?; + + // Use the address of `UNIQUE` as part of the window class name to ensure different + // `wgpu` versions use different names. + static UNIQUE: Mutex = Mutex::new(0); + let class_addr: *const _ = &UNIQUE; + let name = format!("wgpu Device Class {:x}\0", class_addr as usize); + let name = CString::from_vec_with_nul(name.into_bytes()).unwrap(); + + // The window class may already be registered if we are a dynamic library that got + // unloaded & loaded back into the same process. If so, just skip creation. + let already_exists = unsafe { + let mut wc = mem::zeroed::(); + WindowsAndMessaging::GetClassInfoExA( + Some(instance.into()), + PCSTR(name.as_ptr().cast()), + &mut wc, + ) + .is_ok() + }; + if already_exists { + return Ok(name); + } + + // Use a wrapper function for compatibility with `windows-rs`. + unsafe extern "system" fn wnd_proc( + window: Foundation::HWND, + msg: u32, + wparam: Foundation::WPARAM, + lparam: Foundation::LPARAM, + ) -> Foundation::LRESULT { + unsafe { WindowsAndMessaging::DefWindowProcA(window, msg, wparam, lparam) } + } + + let window_class = WindowsAndMessaging::WNDCLASSEXA { + cbSize: size_of::() as u32, + style: WindowsAndMessaging::CS_OWNDC, + lpfnWndProc: Some(wnd_proc), + cbClsExtra: 0, + cbWndExtra: 0, + hInstance: instance.into(), + hIcon: WindowsAndMessaging::HICON::default(), + hCursor: WindowsAndMessaging::HCURSOR::default(), + hbrBackground: Gdi::HBRUSH::default(), + lpszMenuName: PCSTR::null(), + lpszClassName: PCSTR(name.as_ptr().cast()), + hIconSm: WindowsAndMessaging::HICON::default(), + }; + + let atom = unsafe { WindowsAndMessaging::RegisterClassExA(&window_class) }; + + if atom == 0 { + return Err(crate::InstanceError::with_source( + String::from("unable to register window class"), + Error::from_thread(), + )); + } + + // We intentionally leak the window class as we only need one per process. + + Ok(name) +} + +fn get_global_window_class() -> Result { + static GLOBAL: LazyLock> = + LazyLock::new(create_global_window_class); + GLOBAL.clone() +} + +struct InstanceDevice { + dc: Gdi::HDC, + + /// This is used to keep the thread owning `dc` alive until this struct is dropped. + _tx: SyncSender<()>, +} + +fn create_instance_device() -> Result { + #[derive(Clone, Copy)] + // TODO: We can get these SendSync definitions in the upstream metadata if this is the case + struct SendDc(Gdi::HDC); + unsafe impl Sync for SendDc {} + unsafe impl Send for SendDc {} + + struct Window { + window: Foundation::HWND, + } + impl Drop for Window { + fn drop(&mut self) { + if let Err(e) = unsafe { WindowsAndMessaging::DestroyWindow(self.window) } { + log::error!("failed to destroy window: {e}"); + } + } + } + + let window_class = get_global_window_class()?; + + let (drop_tx, drop_rx) = sync_channel(0); + let (setup_tx, setup_rx) = sync_channel(0); + + // We spawn a thread which owns the hidden window for this instance. + thread::Builder::new() + .stack_size(256 * 1024) + .name("wgpu-hal WGL Instance Thread".to_owned()) + .spawn(move || { + let setup = (|| { + let instance = unsafe { LibraryLoader::GetModuleHandleA(None) }.map_err(|e| { + crate::InstanceError::with_source( + String::from("unable to get executable instance"), + e, + ) + })?; + + // Create a hidden window since we don't pass `WS_VISIBLE`. + let window = unsafe { + WindowsAndMessaging::CreateWindowExA( + WindowsAndMessaging::WINDOW_EX_STYLE::default(), + PCSTR(window_class.as_ptr().cast()), + PCSTR(window_class.as_ptr().cast()), + WindowsAndMessaging::WINDOW_STYLE::default(), + 0, + 0, + 1, + 1, + None, + None, + Some(instance.into()), + None, + ) + } + .map_err(|e| { + crate::InstanceError::with_source( + String::from("unable to create hidden instance window"), + e, + ) + })?; + let window = Window { window }; + + let dc = unsafe { Gdi::GetDC(Some(window.window)) }; + if dc.is_invalid() { + return Err(crate::InstanceError::with_source( + String::from("unable to create memory device"), + Error::from_thread(), + )); + } + let dc = DeviceContextHandle { + device: dc, + window: window.window, + }; + unsafe { setup_pixel_format(dc.device)? }; + + Ok((window, dc)) + })(); + + match setup { + Ok((_window, dc)) => { + setup_tx.send(Ok(SendDc(dc.device))).unwrap(); + // Wait for the shutdown event to free the window and device context handle. + drop_rx.recv().ok(); + } + Err(err) => { + setup_tx.send(Err(err)).unwrap(); + } + } + }) + .map_err(|e| { + crate::InstanceError::with_source(String::from("unable to create instance thread"), e) + })?; + + let dc = setup_rx.recv().unwrap()?.0; + + Ok(InstanceDevice { dc, _tx: drop_tx }) +} + +impl crate::Instance for Instance { + type A = super::Api; + + unsafe fn init(desc: &crate::InstanceDescriptor<'_>) -> Result { + profiling::scope!("Init OpenGL (WGL) Backend"); + let opengl_module = + unsafe { LibraryLoader::LoadLibraryA(PCSTR(c"opengl32.dll".as_ptr().cast())) } + .map_err(|e| { + crate::InstanceError::with_source( + String::from("unable to load the OpenGL library"), + e, + ) + })?; + + let device = create_instance_device()?; + let dc = device.dc; + + let context = unsafe { OpenGL::wglCreateContext(dc) }.map_err(|e| { + crate::InstanceError::with_source( + String::from("unable to create initial OpenGL context"), + e, + ) + })?; + let context = WglContext { context }; + context.make_current(dc).map_err(|e| { + crate::InstanceError::with_source( + String::from("unable to set initial OpenGL context as current"), + e, + ) + })?; + + let extra = Wgl::load_with(|name| load_gl_func(name, None)); + let extensions = get_extensions(&extra, dc); + + let can_use_profile = extensions.contains("WGL_ARB_create_context_profile") + && extra.CreateContextAttribsARB.is_loaded(); + + let context = if can_use_profile { + let attributes = [ + CONTEXT_PROFILE_MASK_ARB as c_int, + CONTEXT_CORE_PROFILE_BIT_ARB as c_int, + CONTEXT_FLAGS_ARB as c_int, + if desc.flags.contains(InstanceFlags::DEBUG) { + CONTEXT_DEBUG_BIT_ARB as c_int + } else { + 0 + }, + 0, // End of list + ]; + let context = + unsafe { extra.CreateContextAttribsARB(dc.0, ptr::null(), attributes.as_ptr()) }; + if context.is_null() { + return Err(crate::InstanceError::with_source( + String::from("unable to create OpenGL context"), + Error::from_thread(), + )); + } + WglContext { + context: OpenGL::HGLRC(context.cast_mut()), + } + } else { + context + }; + + context.make_current(dc).map_err(|e| { + crate::InstanceError::with_source( + String::from("unable to set OpenGL context as current"), + e, + ) + })?; + + let mut gl = unsafe { + glow::Context::from_loader_function(|name| load_gl_func(name, Some(opengl_module))) + }; + + let extra = Wgl::load_with(|name| load_gl_func(name, None)); + let extensions = get_extensions(&extra, dc); + + let srgb_capable = extensions.contains("WGL_EXT_framebuffer_sRGB") + || extensions.contains("WGL_ARB_framebuffer_sRGB") + || gl + .supported_extensions() + .contains("GL_ARB_framebuffer_sRGB"); + + // In contrast to OpenGL ES, OpenGL requires explicitly enabling sRGB conversions, + // as otherwise the user has to do the sRGB conversion. + if srgb_capable { + unsafe { gl.enable(glow::FRAMEBUFFER_SRGB) }; + } + + if desc.flags.contains(InstanceFlags::VALIDATION) && gl.supports_debug() { + log::debug!("Enabling GL debug output"); + unsafe { gl.enable(glow::DEBUG_OUTPUT) }; + unsafe { gl.debug_message_callback(super::gl_debug_message_callback) }; + } + + // Wrap in ManuallyDrop to make it easier to "current" the GL context before dropping this + // GLOW context, which could also happen if a panic occurs after we uncurrent the context + // below but before Inner is constructed. + let gl = ManuallyDrop::new(gl); + context.unmake_current().map_err(|e| { + crate::InstanceError::with_source( + String::from("unable to unset the current WGL context"), + e, + ) + })?; + + Ok(Instance { + inner: Arc::new(Mutex::new(Inner { + device, + gl, + context: Some(context), + })), + options: desc.backend_options.gl.clone(), + srgb_capable, + }) + } + + unsafe fn create_surface( + &self, + display_handle: RawDisplayHandle, + window_handle: RawWindowHandle, + ) -> Result { + assert!(matches!(display_handle, RawDisplayHandle::Windows(_))); + let window = if let RawWindowHandle::Win32(handle) = window_handle { + handle + } else { + return Err(crate::InstanceError::new(format!( + "unsupported window: {window_handle:?}" + ))); + }; + Ok(Surface { + // This cast exists because of https://github.com/rust-windowing/raw-window-handle/issues/171 + window: Foundation::HWND(window.hwnd.get() as *mut _), + presentable: true, + swapchain: RwLock::new(None), + srgb_capable: self.srgb_capable, + }) + } + + unsafe fn enumerate_adapters( + &self, + _surface_hint: Option<&Surface>, + ) -> Vec> { + unsafe { + super::Adapter::expose( + AdapterContext { + inner: self.inner.clone(), + }, + self.options.clone(), + ) + } + .into_iter() + .collect() + } +} + +impl super::Adapter { + /// Creates a new external adapter using the specified loader function. + /// + /// # Safety + /// + /// - The underlying OpenGL ES context must be current. + /// - The underlying OpenGL ES context must be current when interfacing with any objects returned by + /// wgpu-hal from this adapter. + /// - The underlying OpenGL ES context must be current when dropping this adapter and when + /// dropping any objects returned from this adapter. + pub unsafe fn new_external( + fun: impl FnMut(&str) -> *const c_void, + options: wgt::GlBackendOptions, + ) -> Option> { + let context = unsafe { glow::Context::from_loader_function(fun) }; + unsafe { + Self::expose( + AdapterContext { + inner: Arc::new(Mutex::new(Inner { + gl: ManuallyDrop::new(context), + device: create_instance_device().ok()?, + context: None, + })), + }, + options, + ) + } + } + + pub fn adapter_context(&self) -> &AdapterContext { + &self.shared.context + } +} + +impl super::Device { + /// Returns the underlying WGL context. + pub fn context(&self) -> &AdapterContext { + &self.shared.context + } +} + +struct DeviceContextHandle { + device: Gdi::HDC, + window: Foundation::HWND, +} + +impl Drop for DeviceContextHandle { + fn drop(&mut self) { + unsafe { + Gdi::ReleaseDC(Some(self.window), self.device); + }; + } +} + +pub struct Swapchain { + framebuffer: glow::Framebuffer, + renderbuffer: glow::Renderbuffer, + + /// Extent because the window lies + extent: wgt::Extent3d, + + format: wgt::TextureFormat, + format_desc: super::TextureFormatDesc, + #[allow(unused)] + sample_type: wgt::TextureSampleType, +} + +pub struct Surface { + window: Foundation::HWND, + pub(super) presentable: bool, + swapchain: RwLock>, + srgb_capable: bool, +} + +unsafe impl Send for Surface {} +unsafe impl Sync for Surface {} + +impl Surface { + pub(super) unsafe fn present( + &self, + _suf_texture: super::Texture, + context: &AdapterContext, + ) -> Result<(), crate::SurfaceError> { + let swapchain = self.swapchain.read(); + let sc = swapchain.as_ref().unwrap(); + let dc = unsafe { Gdi::GetDC(Some(self.window)) }; + if dc.is_invalid() { + log::error!( + "unable to get the device context from window: {}", + Error::from_thread() + ); + return Err(crate::SurfaceError::Other( + "unable to get the device context from window", + )); + } + let dc = DeviceContextHandle { + device: dc, + window: self.window, + }; + + let gl = context.lock_with_dc(dc.device).map_err(|e| { + log::error!("unable to make the OpenGL context current for surface: {e}",); + crate::SurfaceError::Other("unable to make the OpenGL context current for surface") + })?; + + unsafe { gl.bind_framebuffer(glow::DRAW_FRAMEBUFFER, None) }; + unsafe { gl.bind_framebuffer(glow::READ_FRAMEBUFFER, Some(sc.framebuffer)) }; + + if self.srgb_capable { + // Disable sRGB conversions for `glBlitFramebuffer` as behavior does diverge between + // drivers and formats otherwise and we want to ensure no sRGB conversions happen. + unsafe { gl.disable(glow::FRAMEBUFFER_SRGB) }; + } + + // Note the Y-flipping here. GL's presentation is not flipped, + // but main rendering is. Therefore, we Y-flip the output positions + // in the shader, and also this blit. + unsafe { + gl.blit_framebuffer( + 0, + sc.extent.height as i32, + sc.extent.width as i32, + 0, + 0, + 0, + sc.extent.width as i32, + sc.extent.height as i32, + glow::COLOR_BUFFER_BIT, + glow::NEAREST, + ) + }; + + if self.srgb_capable { + unsafe { gl.enable(glow::FRAMEBUFFER_SRGB) }; + } + + unsafe { gl.bind_renderbuffer(glow::RENDERBUFFER, None) }; + unsafe { gl.bind_framebuffer(glow::READ_FRAMEBUFFER, None) }; + + if let Err(e) = unsafe { OpenGL::SwapBuffers(dc.device) } { + log::error!("unable to swap buffers: {e}"); + return Err(crate::SurfaceError::Other("unable to swap buffers")); + } + + Ok(()) + } + + pub fn supports_srgb(&self) -> bool { + self.srgb_capable + } +} + +impl crate::Surface for Surface { + type A = super::Api; + + unsafe fn configure( + &self, + device: &super::Device, + config: &crate::SurfaceConfiguration, + ) -> Result<(), crate::SurfaceError> { + // Remove the old configuration. + unsafe { self.unconfigure(device) }; + + let dc = unsafe { Gdi::GetDC(Some(self.window)) }; + if dc.is_invalid() { + log::error!( + "unable to get the device context from window: {}", + Error::from_thread() + ); + return Err(crate::SurfaceError::Other( + "unable to get the device context from window", + )); + } + let dc = DeviceContextHandle { + device: dc, + window: self.window, + }; + + if let Err(e) = unsafe { setup_pixel_format(dc.device) } { + log::error!("unable to setup surface pixel format: {e}",); + return Err(crate::SurfaceError::Other( + "unable to setup surface pixel format", + )); + } + + let format_desc = device.shared.describe_texture_format(config.format); + let gl = &device.shared.context.lock_with_dc(dc.device).map_err(|e| { + log::error!("unable to make the OpenGL context current for surface: {e}",); + crate::SurfaceError::Other("unable to make the OpenGL context current for surface") + })?; + + let renderbuffer = unsafe { gl.create_renderbuffer() }.map_err(|error| { + log::error!("Internal swapchain renderbuffer creation failed: {error}"); + crate::DeviceError::OutOfMemory + })?; + unsafe { gl.bind_renderbuffer(glow::RENDERBUFFER, Some(renderbuffer)) }; + unsafe { + gl.renderbuffer_storage( + glow::RENDERBUFFER, + format_desc.internal, + config.extent.width as _, + config.extent.height as _, + ) + }; + + let framebuffer = unsafe { gl.create_framebuffer() }.map_err(|error| { + log::error!("Internal swapchain framebuffer creation failed: {error}"); + crate::DeviceError::OutOfMemory + })?; + unsafe { gl.bind_framebuffer(glow::READ_FRAMEBUFFER, Some(framebuffer)) }; + unsafe { + gl.framebuffer_renderbuffer( + glow::READ_FRAMEBUFFER, + glow::COLOR_ATTACHMENT0, + glow::RENDERBUFFER, + Some(renderbuffer), + ) + }; + unsafe { gl.bind_renderbuffer(glow::RENDERBUFFER, None) }; + unsafe { gl.bind_framebuffer(glow::READ_FRAMEBUFFER, None) }; + + // Setup presentation mode + let extra = Wgl::load_with(|name| load_gl_func(name, None)); + let extensions = get_extensions(&extra, dc.device); + if !(extensions.contains("WGL_EXT_swap_control") && extra.SwapIntervalEXT.is_loaded()) { + log::error!("WGL_EXT_swap_control is unsupported"); + return Err(crate::SurfaceError::Other( + "WGL_EXT_swap_control is unsupported", + )); + } + + let vsync = match config.present_mode { + wgt::PresentMode::Immediate => false, + wgt::PresentMode::Fifo => true, + _ => { + log::error!("unsupported present mode: {:?}", config.present_mode); + return Err(crate::SurfaceError::Other("unsupported present mode")); + } + }; + + if unsafe { extra.SwapIntervalEXT(if vsync { 1 } else { 0 }) } == Foundation::FALSE.0 { + log::error!("unable to set swap interval: {}", Error::from_thread()); + return Err(crate::SurfaceError::Other("unable to set swap interval")); + } + + self.swapchain.write().replace(Swapchain { + renderbuffer, + framebuffer, + extent: config.extent, + format: config.format, + format_desc, + sample_type: wgt::TextureSampleType::Float { filterable: false }, + }); + + Ok(()) + } + + unsafe fn unconfigure(&self, device: &super::Device) { + let gl = &device.shared.context.lock(); + if let Some(sc) = self.swapchain.write().take() { + unsafe { + gl.delete_renderbuffer(sc.renderbuffer); + gl.delete_framebuffer(sc.framebuffer) + }; + } + } + + unsafe fn acquire_texture( + &self, + _timeout_ms: Option, + _fence: &super::Fence, + ) -> Result, crate::SurfaceError> { + let swapchain = self.swapchain.read(); + let sc = swapchain.as_ref().unwrap(); + let texture = super::Texture { + inner: super::TextureInner::Renderbuffer { + raw: sc.renderbuffer, + }, + drop_guard: None, + array_layer_count: 1, + mip_level_count: 1, + format: sc.format, + format_desc: sc.format_desc.clone(), + copy_size: crate::CopyExtent { + width: sc.extent.width, + height: sc.extent.height, + depth: 1, + }, + }; + Ok(crate::AcquiredSurfaceTexture { + texture, + suboptimal: false, + }) + } + unsafe fn discard_texture(&self, _texture: super::Texture) {} +} diff --git a/third_party/wgpu-hal-29.0.4/src/lib.rs b/third_party/wgpu-hal-29.0.4/src/lib.rs new file mode 100644 index 0000000..3f36e67 --- /dev/null +++ b/third_party/wgpu-hal-29.0.4/src/lib.rs @@ -0,0 +1,2852 @@ +//! A cross-platform unsafe graphics abstraction. +//! +//! This crate defines a set of traits abstracting over modern graphics APIs, +//! with implementations ("backends") for Vulkan, Metal, Direct3D, and GL. +//! +//! `wgpu-hal` is a spiritual successor to +//! [gfx-hal](https://github.com/gfx-rs/gfx), but with reduced scope, and +//! oriented towards WebGPU implementation goals. It has no overhead for +//! validation or tracking, and the API translation overhead is kept to the bare +//! minimum by the design of WebGPU. This API can be used for resource-demanding +//! applications and engines. +//! +//! The `wgpu-hal` crate's main design choices: +//! +//! - Our traits are meant to be *portable*: proper use +//! should get equivalent results regardless of the backend. +//! +//! - Our traits' contracts are *unsafe*: implementations perform minimal +//! validation, if any, and incorrect use will often cause undefined behavior. +//! This allows us to minimize the overhead we impose over the underlying +//! graphics system. If you need safety, the [`wgpu-core`] crate provides a +//! safe API for driving `wgpu-hal`, implementing all necessary validation, +//! resource state tracking, and so on. (Note that `wgpu-core` is designed for +//! use via FFI; the [`wgpu`] crate provides more idiomatic Rust bindings for +//! `wgpu-core`.) Or, you can do your own validation. +//! +//! - In the same vein, returned errors *only cover cases the user can't +//! anticipate*, like running out of memory or losing the device. Any errors +//! that the user could reasonably anticipate are their responsibility to +//! avoid. For example, `wgpu-hal` returns no error for mapping a buffer that's +//! not mappable: as the buffer creator, the user should already know if they +//! can map it. +//! +//! - We use *static dispatch*. The traits are not +//! generally object-safe. You must select a specific backend type +//! like [`vulkan::Api`] or [`metal::Api`], and then use that +//! according to the main traits, or call backend-specific methods. +//! +//! - We use *idiomatic Rust parameter passing*, +//! taking objects by reference, returning them by value, and so on, +//! unlike `wgpu-core`, which refers to objects by ID. +//! +//! - We map buffer contents *persistently*. This means that the buffer can +//! remain mapped on the CPU while the GPU reads or writes to it. You must +//! explicitly indicate when data might need to be transferred between CPU and +//! GPU, if [`Device::map_buffer`] indicates that this is necessary. +//! +//! - You must record *explicit barriers* between different usages of a +//! resource. For example, if a buffer is written to by a compute +//! shader, and then used as and index buffer to a draw call, you +//! must use [`CommandEncoder::transition_buffers`] between those two +//! operations. +//! +//! - Pipeline layouts are *explicitly specified* when setting bind groups. +//! Incompatible layouts disturb groups bound at higher indices. +//! +//! - The API *accepts collections as iterators*, to avoid forcing the user to +//! store data in particular containers. The implementation doesn't guarantee +//! that any of the iterators are drained, unless stated otherwise by the +//! function documentation. For this reason, we recommend that iterators don't +//! do any mutating work. +//! +//! Unfortunately, `wgpu-hal`'s safety requirements are not fully documented. +//! Ideally, all trait methods would have doc comments setting out the +//! requirements users must meet to ensure correct and portable behavior. If you +//! are aware of a specific requirement that a backend imposes that is not +//! ensured by the traits' documented rules, please file an issue. Or, if you are +//! a capable technical writer, please file a pull request! +//! +//! [`wgpu-core`]: https://crates.io/crates/wgpu-core +//! [`wgpu`]: https://crates.io/crates/wgpu +//! [`vulkan::Api`]: vulkan/struct.Api.html +//! [`metal::Api`]: metal/struct.Api.html +//! +//! ## Primary backends +//! +//! The `wgpu-hal` crate has full-featured backends implemented on the following +//! platform graphics APIs: +//! +//! - Vulkan, available on Linux, Android, and Windows, using the [`ash`] crate's +//! Vulkan bindings. It's also available on macOS, if you install [MoltenVK]. +//! +//! - Metal on macOS, using the [`metal`] crate's bindings. +//! +//! - Direct3D 12 on Windows, using the [`windows`] crate's bindings. +//! +//! [`ash`]: https://crates.io/crates/ash +//! [MoltenVK]: https://github.com/KhronosGroup/MoltenVK +//! [`metal`]: https://crates.io/crates/metal +//! [`windows`]: https://crates.io/crates/windows +//! +//! ## Secondary backends +//! +//! The `wgpu-hal` crate has a partial implementation based on the following +//! platform graphics API: +//! +//! - The GL backend is available anywhere OpenGL, OpenGL ES, or WebGL are +//! available. See the [`gles`] module documentation for details. +//! +//! [`gles`]: gles/index.html +//! +//! You can see what capabilities an adapter is missing by checking the +//! [`DownlevelCapabilities`][tdc] in [`ExposedAdapter::capabilities`], available +//! from [`Instance::enumerate_adapters`]. +//! +//! The API is generally designed to fit the primary backends better than the +//! secondary backends, so the latter may impose more overhead. +//! +//! [tdc]: wgt::DownlevelCapabilities +//! +//! ## Traits +//! +//! The `wgpu-hal` crate defines a handful of traits that together +//! represent a cross-platform abstraction for modern GPU APIs. +//! +//! - The [`Api`] trait represents a `wgpu-hal` backend. It has no methods of its +//! own, only a collection of associated types. +//! +//! - [`Api::Instance`] implements the [`Instance`] trait. [`Instance::init`] +//! creates an instance value, which you can use to enumerate the adapters +//! available on the system. For example, [`vulkan::Api::Instance::init`][Ii] +//! returns an instance that can enumerate the Vulkan physical devices on your +//! system. +//! +//! - [`Api::Adapter`] implements the [`Adapter`] trait, representing a +//! particular device from a particular backend. For example, a Vulkan instance +//! might have a Lavapipe software adapter and a GPU-based adapter. +//! +//! - [`Api::Device`] implements the [`Device`] trait, representing an active +//! link to a device. You get a device value by calling [`Adapter::open`], and +//! then use it to create buffers, textures, shader modules, and so on. +//! +//! - [`Api::Queue`] implements the [`Queue`] trait, which you use to submit +//! command buffers to a given device. +//! +//! - [`Api::CommandEncoder`] implements the [`CommandEncoder`] trait, which you +//! use to build buffers of commands to submit to a queue. This has all the +//! methods for drawing and running compute shaders, which is presumably what +//! you're here for. +//! +//! - [`Api::Surface`] implements the [`Surface`] trait, which represents a +//! swapchain for presenting images on the screen, via interaction with the +//! system's window manager. +//! +//! The [`Api`] trait has various other associated types like [`Api::Buffer`] and +//! [`Api::Texture`] that represent resources the rest of the interface can +//! operate on, but these generally do not have their own traits. +//! +//! [Ii]: Instance::init +//! +//! ## Validation is the calling code's responsibility, not `wgpu-hal`'s +//! +//! As much as possible, `wgpu-hal` traits place the burden of validation, +//! resource tracking, and state tracking on the caller, not on the trait +//! implementations themselves. Anything which can reasonably be handled in +//! backend-independent code should be. A `wgpu_hal` backend's sole obligation is +//! to provide portable behavior, and report conditions that the calling code +//! can't reasonably anticipate, like device loss or running out of memory. +//! +//! The `wgpu` crate collection is intended for use in security-sensitive +//! applications, like web browsers, where the API is available to untrusted +//! code. This means that `wgpu-core`'s validation is not simply a service to +//! developers, to be provided opportunistically when the performance costs are +//! acceptable and the necessary data is ready at hand. Rather, `wgpu-core`'s +//! validation must be exhaustive, to ensure that even malicious content cannot +//! provoke and exploit undefined behavior in the platform's graphics API. +//! +//! Because graphics APIs' requirements are complex, the only practical way for +//! `wgpu` to provide exhaustive validation is to comprehensively track the +//! lifetime and state of all the resources in the system. Implementing this +//! separately for each backend is infeasible; effort would be better spent +//! making the cross-platform validation in `wgpu-core` legible and trustworthy. +//! Fortunately, the requirements are largely similar across the various +//! platforms, so cross-platform validation is practical. +//! +//! Some backends have specific requirements that aren't practical to foist off +//! on the `wgpu-hal` user. For example, properly managing macOS Objective-C or +//! Microsoft COM reference counts is best handled by using appropriate pointer +//! types within the backend. +//! +//! A desire for "defense in depth" may suggest performing additional validation +//! in `wgpu-hal` when the opportunity arises, but this must be done with +//! caution. Even experienced contributors infer the expectations their changes +//! must meet by considering not just requirements made explicit in types, tests, +//! assertions, and comments, but also those implicit in the surrounding code. +//! When one sees validation or state-tracking code in `wgpu-hal`, it is tempting +//! to conclude, "Oh, `wgpu-hal` checks for this, so `wgpu-core` needn't worry +//! about it - that would be redundant!" The responsibility for exhaustive +//! validation always rests with `wgpu-core`, regardless of what may or may not +//! be checked in `wgpu-hal`. +//! +//! To this end, any "defense in depth" validation that does appear in `wgpu-hal` +//! for requirements that `wgpu-core` should have enforced should report failure +//! via the `unreachable!` macro, because problems detected at this stage always +//! indicate a bug in `wgpu-core`. +//! +//! ## Debugging +//! +//! Most of the information on the wiki [Debugging wgpu Applications][wiki-debug] +//! page still applies to this API, with the exception of API tracing/replay +//! functionality, which is only available in `wgpu-core`. +//! +//! [wiki-debug]: https://github.com/gfx-rs/wgpu/wiki/Debugging-wgpu-Applications + +#![no_std] +#![cfg_attr(docsrs, feature(doc_cfg))] +#![allow( + // this happens on the GL backend, where it is both thread safe and non-thread safe in the same code. + clippy::arc_with_non_send_sync, + // We don't use syntax sugar where it's not necessary. + clippy::match_like_matches_macro, + // Redundant matching is more explicit. + clippy::redundant_pattern_matching, + // Explicit lifetimes are often easier to reason about. + clippy::needless_lifetimes, + // No need for defaults in the internal types. + clippy::new_without_default, + // Matches are good and extendable, no need to make an exception here. + clippy::single_match, + // Push commands are more regular than macros. + clippy::vec_init_then_push, + // We unsafe impl `Send` for a reason. + clippy::non_send_fields_in_send_ty, + // TODO! + clippy::missing_safety_doc, + // It gets in the way a lot and does not prevent bugs in practice. + clippy::pattern_type_mismatch, + // We should investigate these. + clippy::large_enum_variant +)] +#![warn( + clippy::alloc_instead_of_core, + clippy::ptr_as_ptr, + clippy::std_instead_of_alloc, + clippy::std_instead_of_core, + trivial_casts, + trivial_numeric_casts, + unsafe_op_in_unsafe_fn, + unused_extern_crates, + unused_qualifications +)] + +extern crate alloc; +extern crate wgpu_types as wgt; +// Each of these backends needs `std` in some fashion; usually `std::thread` functions. +#[cfg(any(dx12, gles_with_std, metal, vulkan))] +#[macro_use] +extern crate std; + +/// DirectX12 API internals. +#[cfg(dx12)] +pub mod dx12; +/// GLES API internals. +#[cfg(gles)] +pub mod gles; +/// Metal API internals. +#[cfg(metal)] +pub mod metal; +/// A dummy API implementation. +// TODO(https://github.com/gfx-rs/wgpu/issues/7120): this should have a cfg +pub mod noop; +/// Vulkan API internals. +#[cfg(vulkan)] +pub mod vulkan; + +pub mod auxil; +pub mod api { + #[cfg(dx12)] + pub use super::dx12::Api as Dx12; + #[cfg(gles)] + pub use super::gles::Api as Gles; + #[cfg(metal)] + pub use super::metal::Api as Metal; + pub use super::noop::Api as Noop; + #[cfg(vulkan)] + pub use super::vulkan::Api as Vulkan; +} + +mod dynamic; +#[cfg(feature = "validation_canary")] +mod validation_canary; + +#[cfg(feature = "validation_canary")] +pub use validation_canary::{ValidationCanary, VALIDATION_CANARY}; + +pub(crate) use dynamic::impl_dyn_resource; +pub use dynamic::{ + DynAccelerationStructure, DynAcquiredSurfaceTexture, DynAdapter, DynBindGroup, + DynBindGroupLayout, DynBuffer, DynCommandBuffer, DynCommandEncoder, DynComputePipeline, + DynDevice, DynExposedAdapter, DynFence, DynInstance, DynOpenDevice, DynPipelineCache, + DynPipelineLayout, DynQuerySet, DynQueue, DynRenderPipeline, DynResource, DynSampler, + DynShaderModule, DynSurface, DynSurfaceTexture, DynTexture, DynTextureView, +}; + +#[allow(unused)] +use alloc::boxed::Box; +use alloc::{borrow::Cow, string::String, vec::Vec}; +use core::{ + borrow::Borrow, + error::Error, + fmt, + num::{NonZeroU32, NonZeroU64}, + ops::{Range, RangeInclusive}, + ptr::NonNull, +}; + +use bitflags::bitflags; +use raw_window_handle::DisplayHandle; +use thiserror::Error; +use wgt::WasmNotSendSync; + +cfg_if::cfg_if! { + if #[cfg(supports_ptr_atomics)] { + use alloc::sync::Arc; + } else if #[cfg(feature = "portable-atomic")] { + use portable_atomic_util::Arc; + } +} + +// - Vertex + Fragment +// - Compute +// Task + Mesh + Fragment +pub const MAX_CONCURRENT_SHADER_STAGES: usize = 3; +pub const MAX_ANISOTROPY: u8 = 16; +pub const MAX_BIND_GROUPS: usize = 8; +pub const MAX_VERTEX_BUFFERS: usize = 16; +pub const MAX_COLOR_ATTACHMENTS: usize = 8; +pub const MAX_MIP_LEVELS: u32 = 16; +/// Size of a single occlusion/timestamp query, when copied into a buffer, in bytes. +/// cbindgen:ignore +pub const QUERY_SIZE: wgt::BufferAddress = 8; + +pub type Label<'a> = Option<&'a str>; +pub type MemoryRange = Range; +pub type FenceValue = u64; +#[cfg(supports_64bit_atomics)] +pub type AtomicFenceValue = core::sync::atomic::AtomicU64; +#[cfg(not(supports_64bit_atomics))] +pub type AtomicFenceValue = portable_atomic::AtomicU64; + +/// A callback to signal that wgpu is no longer using a resource. +#[cfg(any(gles, vulkan))] +pub type DropCallback = Box; + +#[cfg(any(gles, vulkan))] +pub struct DropGuard { + callback: Option, +} + +#[cfg(all(any(gles, vulkan), any(native, Emscripten)))] +impl DropGuard { + fn from_option(callback: Option) -> Option { + callback.map(|callback| Self { + callback: Some(callback), + }) + } +} + +#[cfg(any(gles, vulkan))] +impl Drop for DropGuard { + fn drop(&mut self) { + if let Some(cb) = self.callback.take() { + (cb)(); + } + } +} + +#[cfg(any(gles, vulkan))] +impl fmt::Debug for DropGuard { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("DropGuard").finish() + } +} + +#[derive(Clone, Debug, PartialEq, Eq, Error)] +pub enum DeviceError { + #[error("Out of memory")] + OutOfMemory, + #[error("Device is lost")] + Lost, + #[error("Unexpected error variant (driver implementation is at fault)")] + Unexpected, +} + +#[cfg(any(dx12, vulkan))] +impl From for DeviceError { + fn from(result: gpu_allocator::AllocationError) -> Self { + match result { + gpu_allocator::AllocationError::OutOfMemory => Self::OutOfMemory, + gpu_allocator::AllocationError::FailedToMap(e) => { + log::error!("gpu-allocator: Failed to map: {e}"); + Self::Lost + } + gpu_allocator::AllocationError::NoCompatibleMemoryTypeFound => { + log::error!("gpu-allocator: No Compatible Memory Type Found"); + Self::Lost + } + gpu_allocator::AllocationError::InvalidAllocationCreateDesc => { + log::error!("gpu-allocator: Invalid Allocation Creation Description"); + Self::Lost + } + gpu_allocator::AllocationError::InvalidAllocatorCreateDesc(e) => { + log::error!("gpu-allocator: Invalid Allocator Creation Description: {e}"); + Self::Lost + } + + gpu_allocator::AllocationError::Internal(e) => { + log::error!("gpu-allocator: Internal Error: {e}"); + Self::Lost + } + gpu_allocator::AllocationError::BarrierLayoutNeedsDevice10 + | gpu_allocator::AllocationError::CastableFormatsRequiresEnhancedBarriers + | gpu_allocator::AllocationError::CastableFormatsRequiresAtLeastDevice12 => { + unreachable!() + } + } + } +} + +// A copy of gpu_allocator::AllocationSizes, allowing to read the configured value for +// the dx12 backend, we should instead add getters to gpu_allocator::AllocationSizes +// and remove this type. +// https://github.com/Traverse-Research/gpu-allocator/issues/295 +#[cfg_attr(not(any(dx12, vulkan)), expect(dead_code))] +pub(crate) struct AllocationSizes { + pub(crate) min_device_memblock_size: u64, + pub(crate) max_device_memblock_size: u64, + pub(crate) min_host_memblock_size: u64, + pub(crate) max_host_memblock_size: u64, +} + +impl AllocationSizes { + #[allow(dead_code)] // may be unused on some platforms + pub(crate) fn from_memory_hints(memory_hints: &wgt::MemoryHints) -> Self { + // TODO: the allocator's configuration should take hardware capability into + // account. + const MB: u64 = 1024 * 1024; + + match memory_hints { + wgt::MemoryHints::Performance => Self { + min_device_memblock_size: 128 * MB, + max_device_memblock_size: 256 * MB, + min_host_memblock_size: 64 * MB, + max_host_memblock_size: 128 * MB, + }, + wgt::MemoryHints::MemoryUsage => Self { + min_device_memblock_size: 8 * MB, + max_device_memblock_size: 64 * MB, + min_host_memblock_size: 4 * MB, + max_host_memblock_size: 32 * MB, + }, + wgt::MemoryHints::Manual { + suballocated_device_memory_block_size, + } => { + // TODO: https://github.com/gfx-rs/wgpu/issues/8625 + // Would it be useful to expose the host size in memory hints + // instead of always using half of the device size? + let device_size = suballocated_device_memory_block_size; + let host_size = device_size.start / 2..device_size.end / 2; + + // gpu_allocator clamps the sizes between 4MiB and 256MiB, but we clamp them ourselves since we use + // the sizes when detecting high memory pressure and there is no way to query the values otherwise. + Self { + min_device_memblock_size: device_size.start.clamp(4 * MB, 256 * MB), + max_device_memblock_size: device_size.end.clamp(4 * MB, 256 * MB), + min_host_memblock_size: host_size.start.clamp(4 * MB, 256 * MB), + max_host_memblock_size: host_size.end.clamp(4 * MB, 256 * MB), + } + } + } + } +} + +#[cfg(any(dx12, vulkan))] +impl From for gpu_allocator::AllocationSizes { + fn from(value: AllocationSizes) -> gpu_allocator::AllocationSizes { + gpu_allocator::AllocationSizes::new( + value.min_device_memblock_size, + value.min_host_memblock_size, + ) + .with_max_device_memblock_size(value.max_device_memblock_size) + .with_max_host_memblock_size(value.max_host_memblock_size) + } +} + +#[allow(dead_code)] // may be unused on some platforms +#[cold] +fn hal_usage_error(txt: T) -> ! { + panic!("wgpu-hal invariant was violated (usage error): {txt}") +} + +#[allow(dead_code)] // may be unused on some platforms +#[cold] +fn hal_internal_error(txt: T) -> ! { + panic!("wgpu-hal ran into a preventable internal error: {txt}") +} + +#[derive(Clone, Debug, Eq, PartialEq, Error)] +pub enum ShaderError { + #[error("Compilation failed: {0:?}")] + Compilation(String), + #[error(transparent)] + Device(#[from] DeviceError), +} + +#[derive(Clone, Debug, Eq, PartialEq, Error)] +pub enum PipelineError { + #[error("Linkage failed for stage {0:?}: {1}")] + Linkage(wgt::ShaderStages, String), + #[error("Entry point for stage {0:?} is invalid")] + EntryPoint(naga::ShaderStage), + #[error(transparent)] + Device(#[from] DeviceError), + #[error("Pipeline constant error for stage {0:?}: {1}")] + PipelineConstants(wgt::ShaderStages, String), +} + +#[derive(Clone, Debug, Eq, PartialEq, Error)] +pub enum PipelineCacheError { + #[error(transparent)] + Device(#[from] DeviceError), +} + +#[derive(Clone, Debug, Eq, PartialEq, Error)] +pub enum SurfaceError { + #[error("Surface is lost")] + Lost, + #[error("Surface is outdated, needs to be re-created")] + Outdated, + #[error("Timed out waiting for a surface texture")] + Timeout, + #[error("The window is occluded (e.g. minimized or behind another window). Try again once the window is no longer occluded.")] + Occluded, + #[error(transparent)] + Device(#[from] DeviceError), + #[error("Other reason: {0}")] + Other(&'static str), +} + +/// Error occurring while trying to create an instance, or create a surface from an instance; +/// typically relating to the state of the underlying graphics API or hardware. +#[derive(Clone, Debug, Error)] +#[error("{message}")] +pub struct InstanceError { + /// These errors are very platform specific, so do not attempt to encode them as an enum. + /// + /// This message should describe the problem in sufficient detail to be useful for a + /// user-to-developer “why won't this work on my machine” bug report, and otherwise follow + /// . + message: String, + + /// Underlying error value, if any is available. + #[source] + source: Option>, +} + +impl InstanceError { + #[allow(dead_code)] // may be unused on some platforms + pub(crate) fn new(message: String) -> Self { + Self { + message, + source: None, + } + } + #[allow(dead_code)] // may be unused on some platforms + pub(crate) fn with_source(message: String, source: impl Error + Send + Sync + 'static) -> Self { + cfg_if::cfg_if! { + if #[cfg(supports_ptr_atomics)] { + let source = Arc::new(source); + } else { + // TODO(https://github.com/rust-lang/rust/issues/18598): avoid indirection via Box once arbitrary types support unsized coercion + let source: Box = Box::new(source); + let source = Arc::from(source); + } + } + Self { + message, + source: Some(source), + } + } +} + +/// All the types and methods that make up a implementation on top of a backend. +/// +/// Only the types that have non-dyn trait bounds have methods on them. Most methods +/// are either on [`CommandEncoder`] or [`Device`]. +/// +/// The api can either be used through generics (through use of this trait and associated +/// types) or dynamically through using the `Dyn*` traits. +pub trait Api: Clone + fmt::Debug + Sized + WasmNotSendSync + 'static { + const VARIANT: wgt::Backend; + + type Instance: DynInstance + Instance; + type Surface: DynSurface + Surface; + type Adapter: DynAdapter + Adapter; + type Device: DynDevice + Device; + + type Queue: DynQueue + Queue; + type CommandEncoder: DynCommandEncoder + CommandEncoder; + + /// This API's command buffer type. + /// + /// The only thing you can do with `CommandBuffer`s is build them + /// with a [`CommandEncoder`] and then pass them to + /// [`Queue::submit`] for execution, or destroy them by passing + /// them to [`CommandEncoder::reset_all`]. + /// + /// [`CommandEncoder`]: Api::CommandEncoder + type CommandBuffer: DynCommandBuffer; + + type Buffer: DynBuffer; + type Texture: DynTexture; + type SurfaceTexture: DynSurfaceTexture + Borrow; + type TextureView: DynTextureView; + type Sampler: DynSampler; + type QuerySet: DynQuerySet; + + /// A value you can block on to wait for something to finish. + /// + /// A `Fence` holds a monotonically increasing [`FenceValue`]. You can call + /// [`Device::wait`] to block until a fence reaches or passes a value you + /// choose. [`Queue::submit`] can take a `Fence` and a [`FenceValue`] to + /// store in it when the submitted work is complete. + /// + /// Attempting to set a fence to a value less than its current value has no + /// effect. + /// + /// Waiting on a fence returns as soon as the fence reaches *or passes* the + /// requested value. This implies that, in order to reliably determine when + /// an operation has completed, operations must finish in order of + /// increasing fence values: if a higher-valued operation were to finish + /// before a lower-valued operation, then waiting for the fence to reach the + /// lower value could return before the lower-valued operation has actually + /// finished. + type Fence: DynFence; + + type BindGroupLayout: DynBindGroupLayout; + type BindGroup: DynBindGroup; + type PipelineLayout: DynPipelineLayout; + type ShaderModule: DynShaderModule; + type RenderPipeline: DynRenderPipeline; + type ComputePipeline: DynComputePipeline; + type PipelineCache: DynPipelineCache; + + type AccelerationStructure: DynAccelerationStructure + 'static; +} + +pub trait Instance: Sized + WasmNotSendSync { + type A: Api; + + unsafe fn init(desc: &InstanceDescriptor<'_>) -> Result; + unsafe fn create_surface( + &self, + display_handle: raw_window_handle::RawDisplayHandle, + window_handle: raw_window_handle::RawWindowHandle, + ) -> Result<::Surface, InstanceError>; + /// `surface_hint` is only used by the GLES backend targeting WebGL2 + unsafe fn enumerate_adapters( + &self, + surface_hint: Option<&::Surface>, + ) -> Vec>; +} + +pub trait Surface: WasmNotSendSync { + type A: Api; + + /// Configure `self` to use `device`. + /// + /// # Safety + /// + /// - All GPU work using `self` must have been completed. + /// - All [`AcquiredSurfaceTexture`]s must have been destroyed. + /// - All [`Api::TextureView`]s derived from the [`AcquiredSurfaceTexture`]s must have been destroyed. + /// - The surface `self` must not currently be configured to use any other [`Device`]. + unsafe fn configure( + &self, + device: &::Device, + config: &SurfaceConfiguration, + ) -> Result<(), SurfaceError>; + + /// Unconfigure `self` on `device`. + /// + /// # Safety + /// + /// - All GPU work that uses `surface` must have been completed. + /// - All [`AcquiredSurfaceTexture`]s must have been destroyed. + /// - All [`Api::TextureView`]s derived from the [`AcquiredSurfaceTexture`]s must have been destroyed. + /// - The surface `self` must have been configured on `device`. + unsafe fn unconfigure(&self, device: &::Device); + + /// Return the next texture to be presented by `self`, for the caller to draw on. + /// + /// On success, return an [`AcquiredSurfaceTexture`] representing the + /// texture into which the caller should draw the image to be displayed on + /// `self`. + /// + /// If `timeout` elapses before `self` has a texture ready to be acquired, + /// return `Err(SurfaceError::Timeout)`. If `timeout` is `None`, wait + /// indefinitely, with no timeout. + /// + /// # Using an [`AcquiredSurfaceTexture`] + /// + /// On success, this function returns an [`AcquiredSurfaceTexture`] whose + /// [`texture`] field is a [`SurfaceTexture`] from which the caller can + /// [`borrow`] a [`Texture`] to draw on. The [`AcquiredSurfaceTexture`] also + /// carries some metadata about that [`SurfaceTexture`]. + /// + /// All calls to [`Queue::submit`] that draw on that [`Texture`] must also + /// include the [`SurfaceTexture`] in the `surface_textures` argument. + /// + /// When you are done drawing on the texture, you can display it on `self` + /// by passing the [`SurfaceTexture`] and `self` to [`Queue::present`]. + /// + /// If you do not wish to display the texture, you must pass the + /// [`SurfaceTexture`] to [`self.discard_texture`], so that it can be reused + /// by future acquisitions. + /// + /// # Portability + /// + /// Some backends can't support a timeout when acquiring a texture. On these + /// backends, `timeout` is ignored. + /// + /// On macOS, this returns `Err(SurfaceError::Timeout)` when the window is + /// not visible (minimized, fully occluded, or on another virtual desktop) + /// to avoid blocking in `CAMetalLayer.nextDrawable()`. + /// + /// # Safety + /// + /// - The surface `self` must currently be configured on some [`Device`]. + /// + /// - The `fence` argument must be the same [`Fence`] passed to all calls to + /// [`Queue::submit`] that used [`Texture`]s acquired from this surface. + /// + /// - You may only have one texture acquired from `self` at a time. When + /// `acquire_texture` returns `Ok(ast)`, you must pass the returned + /// [`SurfaceTexture`] `ast.texture` to either [`Queue::present`] or + /// [`Surface::discard_texture`] before calling `acquire_texture` again. + /// + /// [`texture`]: AcquiredSurfaceTexture::texture + /// [`SurfaceTexture`]: Api::SurfaceTexture + /// [`borrow`]: alloc::borrow::Borrow::borrow + /// [`Texture`]: Api::Texture + /// [`Fence`]: Api::Fence + /// [`self.discard_texture`]: Surface::discard_texture + unsafe fn acquire_texture( + &self, + timeout: Option, + fence: &::Fence, + ) -> Result, SurfaceError>; + + /// Relinquish an acquired texture without presenting it. + /// + /// After this call, the texture underlying [`SurfaceTexture`] may be + /// returned by subsequent calls to [`self.acquire_texture`]. + /// + /// # Safety + /// + /// - The surface `self` must currently be configured on some [`Device`]. + /// + /// - `texture` must be a [`SurfaceTexture`] returned by a call to + /// [`self.acquire_texture`] that has not yet been passed to + /// [`Queue::present`]. + /// + /// [`SurfaceTexture`]: Api::SurfaceTexture + /// [`self.acquire_texture`]: Surface::acquire_texture + unsafe fn discard_texture(&self, texture: ::SurfaceTexture); +} + +pub trait Adapter: WasmNotSendSync { + type A: Api; + + unsafe fn open( + &self, + features: wgt::Features, + limits: &wgt::Limits, + memory_hints: &wgt::MemoryHints, + ) -> Result, DeviceError>; + + /// Return the set of supported capabilities for a texture format. + unsafe fn texture_format_capabilities( + &self, + format: wgt::TextureFormat, + ) -> TextureFormatCapabilities; + + /// Returns the capabilities of working with a specified surface. + /// + /// `None` means presentation is not supported for it. + unsafe fn surface_capabilities( + &self, + surface: &::Surface, + ) -> Option; + + /// Creates a [`PresentationTimestamp`] using the adapter's WSI. + /// + /// [`PresentationTimestamp`]: wgt::PresentationTimestamp + unsafe fn get_presentation_timestamp(&self) -> wgt::PresentationTimestamp; + + /// The combination of all usages that the are guaranteed to be be ordered by the hardware. + /// If a usage is ordered, then if the buffer state doesn't change between draw calls, + /// there are no barriers needed for synchronization. + fn get_ordered_buffer_usages(&self) -> wgt::BufferUses; + + /// The combination of all usages that the are guaranteed to be be ordered by the hardware. + /// If a usage is ordered, then if the buffer state doesn't change between draw calls, + /// there are no barriers needed for synchronization. + fn get_ordered_texture_usages(&self) -> wgt::TextureUses; +} + +/// A connection to a GPU and a pool of resources to use with it. +/// +/// A `wgpu-hal` `Device` represents an open connection to a specific graphics +/// processor, controlled via the backend [`Device::A`]. A `Device` is mostly +/// used for creating resources. Each `Device` has an associated [`Queue`] used +/// for command submission. +/// +/// On Vulkan a `Device` corresponds to a logical device ([`VkDevice`]). Other +/// backends don't have an exact analog: for example, [`ID3D12Device`]s and +/// [`MTLDevice`]s are owned by the backends' [`wgpu_hal::Adapter`] +/// implementations, and shared by all [`wgpu_hal::Device`]s created from that +/// `Adapter`. +/// +/// A `Device`'s life cycle is generally: +/// +/// 1) Obtain a `Device` and its associated [`Queue`] by calling +/// [`Adapter::open`]. +/// +/// Alternatively, the backend-specific types that implement [`Adapter`] often +/// have methods for creating a `wgpu-hal` `Device` from a platform-specific +/// handle. For example, [`vulkan::Adapter::device_from_raw`] can create a +/// [`vulkan::Device`] from an [`ash::Device`]. +/// +/// 1) Create resources to use on the device by calling methods like +/// [`Device::create_texture`] or [`Device::create_shader_module`]. +/// +/// 1) Call [`Device::create_command_encoder`] to obtain a [`CommandEncoder`], +/// which you can use to build [`CommandBuffer`]s holding commands to be +/// executed on the GPU. +/// +/// 1) Call [`Queue::submit`] on the `Device`'s associated [`Queue`] to submit +/// [`CommandBuffer`]s for execution on the GPU. If needed, call +/// [`Device::wait`] to wait for them to finish execution. +/// +/// 1) Free resources with methods like [`Device::destroy_texture`] or +/// [`Device::destroy_shader_module`]. +/// +/// 1) Drop the device. +/// +/// [`vkDevice`]: https://registry.khronos.org/vulkan/specs/1.3-extensions/html/vkspec.html#VkDevice +/// [`ID3D12Device`]: https://learn.microsoft.com/en-us/windows/win32/api/d3d12/nn-d3d12-id3d12device +/// [`MTLDevice`]: https://developer.apple.com/documentation/metal/mtldevice +/// [`wgpu_hal::Adapter`]: Adapter +/// [`wgpu_hal::Device`]: Device +/// [`vulkan::Adapter::device_from_raw`]: vulkan/struct.Adapter.html#method.device_from_raw +/// [`vulkan::Device`]: vulkan/struct.Device.html +/// [`ash::Device`]: https://docs.rs/ash/latest/ash/struct.Device.html +/// [`CommandBuffer`]: Api::CommandBuffer +/// +/// # Safety +/// +/// As with other `wgpu-hal` APIs, [validation] is the caller's +/// responsibility. Here are the general requirements for all `Device` +/// methods: +/// +/// - Any resource passed to a `Device` method must have been created by that +/// `Device`. For example, a [`Texture`] passed to [`Device::destroy_texture`] must +/// have been created with the `Device` passed as `self`. +/// +/// - Resources may not be destroyed if they are used by any submitted command +/// buffers that have not yet finished execution. +/// +/// [validation]: index.html#validation-is-the-calling-codes-responsibility-not-wgpu-hals +/// [`Texture`]: Api::Texture +pub trait Device: WasmNotSendSync { + type A: Api; + + /// Creates a new buffer. + /// + /// The initial usage is `wgt::BufferUses::empty()`. + unsafe fn create_buffer( + &self, + desc: &BufferDescriptor, + ) -> Result<::Buffer, DeviceError>; + + /// Free `buffer` and any GPU resources it owns. + /// + /// Note that backends are allowed to allocate GPU memory for buffers from + /// allocation pools, and this call is permitted to simply return `buffer`'s + /// storage to that pool, without making it available to other applications. + /// + /// # Safety + /// + /// - The given `buffer` must not currently be mapped. + unsafe fn destroy_buffer(&self, buffer: ::Buffer); + + /// A hook for when a wgpu-core buffer is created from a raw wgpu-hal buffer. + unsafe fn add_raw_buffer(&self, buffer: &::Buffer); + + /// Return a pointer to CPU memory mapping the contents of `buffer`. + /// + /// Buffer mappings are persistent: the buffer may remain mapped on the CPU + /// while the GPU reads or writes to it. (Note that `wgpu_core` does not use + /// this feature: when a `wgpu_core::Buffer` is unmapped, the underlying + /// `wgpu_hal` buffer is also unmapped.) + /// + /// If this function returns `Ok(mapping)`, then: + /// + /// - `mapping.ptr` is the CPU address of the start of the mapped memory. + /// + /// - If `mapping.is_coherent` is `true`, then CPU writes to the mapped + /// memory are immediately visible on the GPU, and vice versa. + /// + /// # Safety + /// + /// - The given `buffer` must have been created with the [`MAP_READ`] or + /// [`MAP_WRITE`] flags set in [`BufferDescriptor::usage`]. + /// + /// - The given `range` must fall within the size of `buffer`. + /// + /// - The caller must avoid data races between the CPU and the GPU. A data + /// race is any pair of accesses to a particular byte, one of which is a + /// write, that are not ordered with respect to each other by some sort of + /// synchronization operation. + /// + /// - If this function returns `Ok(mapping)` and `mapping.is_coherent` is + /// `false`, then: + /// + /// - Every CPU write to a mapped byte followed by a GPU read of that byte + /// must have at least one call to [`Device::flush_mapped_ranges`] + /// covering that byte that occurs between those two accesses. + /// + /// - Every GPU write to a mapped byte followed by a CPU read of that byte + /// must have at least one call to [`Device::invalidate_mapped_ranges`] + /// covering that byte that occurs between those two accesses. + /// + /// Note that the data race rule above requires that all such access pairs + /// be ordered, so it is meaningful to talk about what must occur + /// "between" them. + /// + /// - Zero-sized mappings are not allowed. + /// + /// - The returned [`BufferMapping::ptr`] must not be used after a call to + /// [`Device::unmap_buffer`]. + /// + /// [`MAP_READ`]: wgt::BufferUses::MAP_READ + /// [`MAP_WRITE`]: wgt::BufferUses::MAP_WRITE + unsafe fn map_buffer( + &self, + buffer: &::Buffer, + range: MemoryRange, + ) -> Result; + + /// Remove the mapping established by the last call to [`Device::map_buffer`]. + /// + /// # Safety + /// + /// - The given `buffer` must be currently mapped. + unsafe fn unmap_buffer(&self, buffer: &::Buffer); + + /// Indicate that CPU writes to mapped buffer memory should be made visible to the GPU. + /// + /// # Safety + /// + /// - The given `buffer` must be currently mapped. + /// + /// - All ranges produced by `ranges` must fall within `buffer`'s size. + unsafe fn flush_mapped_ranges(&self, buffer: &::Buffer, ranges: I) + where + I: Iterator; + + /// Indicate that GPU writes to mapped buffer memory should be made visible to the CPU. + /// + /// # Safety + /// + /// - The given `buffer` must be currently mapped. + /// + /// - All ranges produced by `ranges` must fall within `buffer`'s size. + unsafe fn invalidate_mapped_ranges(&self, buffer: &::Buffer, ranges: I) + where + I: Iterator; + + /// Creates a new texture. + /// + /// The initial usage for all subresources is `wgt::TextureUses::UNINITIALIZED`. + unsafe fn create_texture( + &self, + desc: &TextureDescriptor, + ) -> Result<::Texture, DeviceError>; + unsafe fn destroy_texture(&self, texture: ::Texture); + + /// A hook for when a wgpu-core texture is created from a raw wgpu-hal texture. + unsafe fn add_raw_texture(&self, texture: &::Texture); + + unsafe fn create_texture_view( + &self, + texture: &::Texture, + desc: &TextureViewDescriptor, + ) -> Result<::TextureView, DeviceError>; + unsafe fn destroy_texture_view(&self, view: ::TextureView); + unsafe fn create_sampler( + &self, + desc: &SamplerDescriptor, + ) -> Result<::Sampler, DeviceError>; + unsafe fn destroy_sampler(&self, sampler: ::Sampler); + + /// Create a fresh [`CommandEncoder`]. + /// + /// The new `CommandEncoder` is in the "closed" state. + unsafe fn create_command_encoder( + &self, + desc: &CommandEncoderDescriptor<::Queue>, + ) -> Result<::CommandEncoder, DeviceError>; + + /// Creates a bind group layout. + unsafe fn create_bind_group_layout( + &self, + desc: &BindGroupLayoutDescriptor, + ) -> Result<::BindGroupLayout, DeviceError>; + unsafe fn destroy_bind_group_layout(&self, bg_layout: ::BindGroupLayout); + unsafe fn create_pipeline_layout( + &self, + desc: &PipelineLayoutDescriptor<::BindGroupLayout>, + ) -> Result<::PipelineLayout, DeviceError>; + unsafe fn destroy_pipeline_layout(&self, pipeline_layout: ::PipelineLayout); + + #[allow(clippy::type_complexity)] + unsafe fn create_bind_group( + &self, + desc: &BindGroupDescriptor< + ::BindGroupLayout, + ::Buffer, + ::Sampler, + ::TextureView, + ::AccelerationStructure, + >, + ) -> Result<::BindGroup, DeviceError>; + unsafe fn destroy_bind_group(&self, group: ::BindGroup); + + unsafe fn create_shader_module( + &self, + desc: &ShaderModuleDescriptor, + shader: ShaderInput, + ) -> Result<::ShaderModule, ShaderError>; + unsafe fn destroy_shader_module(&self, module: ::ShaderModule); + + #[allow(clippy::type_complexity)] + unsafe fn create_render_pipeline( + &self, + desc: &RenderPipelineDescriptor< + ::PipelineLayout, + ::ShaderModule, + ::PipelineCache, + >, + ) -> Result<::RenderPipeline, PipelineError>; + unsafe fn destroy_render_pipeline(&self, pipeline: ::RenderPipeline); + + #[allow(clippy::type_complexity)] + unsafe fn create_compute_pipeline( + &self, + desc: &ComputePipelineDescriptor< + ::PipelineLayout, + ::ShaderModule, + ::PipelineCache, + >, + ) -> Result<::ComputePipeline, PipelineError>; + unsafe fn destroy_compute_pipeline(&self, pipeline: ::ComputePipeline); + + unsafe fn create_pipeline_cache( + &self, + desc: &PipelineCacheDescriptor<'_>, + ) -> Result<::PipelineCache, PipelineCacheError>; + fn pipeline_cache_validation_key(&self) -> Option<[u8; 16]> { + None + } + unsafe fn destroy_pipeline_cache(&self, cache: ::PipelineCache); + + unsafe fn create_query_set( + &self, + desc: &wgt::QuerySetDescriptor