From 48e99e56b71226ed924cb17800c502d3017d4ceb Mon Sep 17 00:00:00 2001 From: Mike Solar Date: Sat, 12 Sep 2026 20:52:17 +0800 Subject: [PATCH] =?UTF-8?q?render:=20the=20M2=20GPU=20zero-copy=20pipeline?= =?UTF-8?q?=20=E2=80=94=20wgpu=2029,=20shared=20gpui=20device,=20GPU=20col?= =?UTF-8?q?or=20LUTs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit docs/zh/plans/render-pipeline-threads.md M2: the graph's textures stay on the GPU from evaluation through presentation, and presentation runs on the UI's own wgpu device. - wgpu 25 -> 29 (naga 29) across the engine, unifying it with gpui_wgpu so engine textures are directly sampleable by the presenter (a single wgpu remains in the lockfile). - GpuContext::adopt/install_shared: the app registers the window's device at startup and the render thread renders on it; texture_handle hands the raw Arc to SurfaceSource::Texture - zero-copy present on Linux/FreeBSD. The shared slot replaces an engine context that has not touched the GPU yet (startup-order guard) and refuses once it has. - Texture::Gpu shares a GpuLease so clones release the registry token exactly once; the compositor, transitions and adjustment sweeps keep GPU textures end to end (no per-clip readbacks; GPU clears for black/generated frames). - Color management stays on the GPU: the output node + display ICC chain is baked into a 65^3 3D LUT with the exact CPU reference and applied by the present WGSL pass (manual trilinear); ColorTransformJob bakes its OCIO processor the same way. Neither path skips color management. - The explicit readback boundaries accept GPU textures: export encoder, CLI, worker shm, disk cache; CPU OpenFX already read back. - M5 dependency: the YUV->RGB GPU pass (BT.601/709/2020 x limited/full) matches colormath::yuv444p16_to_rgb_f32. - Acceptance: gpu_transfer_counters; single-clip and layered (multi-track + transition + adjustment) playback tests assert zero GPU->CPU readbacks, and the app test asserts adopted-device present is zero-copy. GPU tests hard-fail when OAK_REQUIRE_GPU is set (CI lavapipe) instead of skipping silently. --- .github/workflows/ci.yml | 4 + Cargo.lock | 414 +---- crates/oak-app/src/oakui/gpu.rs | 144 +- crates/oak-app/src/oakui/real.rs | 47 + crates/oak-app/src/oakui/renderops.rs | 133 +- crates/oak-cli/src/engine.rs | 15 +- crates/oak-core/Cargo.toml | 9 +- crates/oak-core/src/backend.rs | 1572 ++++++++++++++++- crates/oak-core/src/colormath.rs | 2 +- crates/oak-core/src/lib.rs | 1 + crates/oak-core/src/lut.rs | 215 +++ crates/oak-core/src/texture.rs | 66 +- crates/oak-render/Cargo.toml | 8 +- crates/oak-render/src/eval.rs | 652 ++++--- crates/oak-render/tests/adjustment_layer.rs | 16 +- crates/oak-render/tests/graph_render.rs | 38 +- crates/oak-render/tests/ofxmisc_blur.rs | 2 +- crates/oak-render/tests/ofxmisc_color.rs | 2 +- crates/oak-render/tests/ofxmisc_gen.rs | 2 +- crates/oak-render/tests/ofxmisc_matrix.rs | 2 +- crates/oak-render/tests/ofxmisc_merge.rs | 2 +- crates/oak-render/tests/pipeline_test.rs | 5 +- .../oak-render/tests/render_threads_test.rs | 284 ++- crates/oak-render/tests/text_outline_glow.rs | 2 +- crates/oak-render/tests/transition_render.rs | 26 +- crates/oak-render/tests/transitionfx.rs | 2 +- crates/oak-task/src/export.rs | 19 +- crates/oak-worker/src/worker.rs | 56 +- docs/zh/plans/render-pipeline-threads.md | 56 +- 29 files changed, 3014 insertions(+), 782 deletions(-) create mode 100644 crates/oak-core/src/lut.rs diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a4450d096..921de9776 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -122,6 +122,10 @@ jobs: # and no failure, so after 1800 s it dumps every hung process's # thread stacks and kills the suite. - name: Test + env: + # lavapipe is present on this job: a missing adapter must fail + # the GPU acceptance tests instead of silently skipping them. + OAK_REQUIRE_GPU: "1" run: | sudo apt-get install -y gdb run_suite() { diff --git a/Cargo.lock b/Cargo.lock index 1015c34d9..722975399 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1407,17 +1407,6 @@ dependencies = [ "objc", ] -[[package]] -name = "codespan-reporting" -version = "0.12.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fe6d2e5af09e8c8ad56c969f2157a3d4238cebc7c55f0a517728c38f7b200f81" -dependencies = [ - "serde", - "termcolor", - "unicode-width", -] - [[package]] name = "codespan-reporting" version = "0.13.1" @@ -1670,7 +1659,7 @@ dependencies = [ "core-graphics2", "io-surface", "libc", - "metal 0.33.0", + "metal", ] [[package]] @@ -2806,18 +2795,6 @@ dependencies = [ "regex-syntax", ] -[[package]] -name = "glow" -version = "0.16.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c5e5ea60d70410161c8bf5da3fdfeaa1c72ed2c15f8bbb9d19fe3a4fad085f08" -dependencies = [ - "js-sys", - "slotmap", - "wasm-bindgen", - "web-sys", -] - [[package]] name = "glow" version = "0.17.0" @@ -2839,37 +2816,6 @@ dependencies = [ "gl_generator", ] -[[package]] -name = "gpu-alloc" -version = "0.6.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "45cf04b2726f02df5508c6de726acdc90cdf97ac771a9a0ffd8ba10a6e696bf9" -dependencies = [ - "bitflags 2.13.1", - "gpu-alloc-types", -] - -[[package]] -name = "gpu-alloc-types" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b2bbed164dd10ed526c2e4fe3e721ca4a71c61730e5aafac6844b417b3227058" -dependencies = [ - "bitflags 2.13.1", -] - -[[package]] -name = "gpu-allocator" -version = "0.27.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c151a2a5ef800297b4e79efa4f4bec035c5f51d5ae587287c9b952bdf734cacd" -dependencies = [ - "log", - "presser", - "thiserror 1.0.69", - "windows 0.58.0", -] - [[package]] name = "gpu-allocator" version = "0.28.0" @@ -2948,7 +2894,7 @@ dependencies = [ "log", "lyon", "mach2 0.5.0", - "metal 0.33.0", + "metal", "num_cpus", "objc", "parking", @@ -3100,7 +3046,7 @@ dependencies = [ "libc", "log", "mach2 0.5.0", - "metal 0.33.0", + "metal", "objc", "objc2-app-kit 0.3.2", "parking_lot", @@ -3133,7 +3079,7 @@ dependencies = [ "core-video", "ctor", "foreign-types", - "metal 0.33.0", + "metal", "objc", ] @@ -3235,7 +3181,7 @@ dependencies = [ "wasm-bindgen", "wasm-bindgen-futures", "web-sys", - "wgpu 29.0.4", + "wgpu", "zed-font-kit", ] @@ -4322,21 +4268,6 @@ dependencies = [ "autocfg", ] -[[package]] -name = "metal" -version = "0.31.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f569fb946490b5743ad69813cb19629130ce9374034abe31614a36402d18f99e" -dependencies = [ - "bitflags 2.13.1", - "block", - "core-graphics-types 0.1.3", - "foreign-types", - "log", - "objc", - "paste", -] - [[package]] name = "metal" version = "0.33.0" @@ -4406,32 +4337,6 @@ dependencies = [ "pxfm", ] -[[package]] -name = "naga" -version = "25.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2b977c445f26e49757f9aca3631c3b8b836942cb278d69a92e7b80d3b24da632" -dependencies = [ - "arrayvec", - "bit-set 0.8.0", - "bitflags 2.13.1", - "cfg_aliases", - "codespan-reporting 0.12.0", - "half", - "hashbrown 0.15.5", - "hexf-parse", - "indexmap", - "log", - "num-traits", - "once_cell", - "pp-rs", - "rustc-hash 1.1.0", - "spirv 0.3.0+sdk-1.3.268.0", - "strum 0.26.3", - "thiserror 2.0.20", - "unicode-ident", -] - [[package]] name = "naga" version = "29.0.4" @@ -4443,7 +4348,7 @@ dependencies = [ "bitflags 2.13.1", "cfg-if", "cfg_aliases", - "codespan-reporting 0.13.1", + "codespan-reporting", "half", "hashbrown 0.16.1", "hexf-parse", @@ -4452,8 +4357,9 @@ dependencies = [ "log", "num-traits", "once_cell", + "pp-rs", "rustc-hash 1.1.0", - "spirv 0.4.0+sdk-1.4.341.0", + "spirv", "thiserror 2.0.20", "unicode-ident", ] @@ -4476,7 +4382,7 @@ dependencies = [ "bitflags 2.13.1", "jni-sys 0.3.1", "log", - "ndk-sys 0.6.0+11769913", + "ndk-sys", "num_enum", "thiserror 1.0.69", ] @@ -4487,15 +4393,6 @@ version = "0.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "27b02d87554356db9e9a873add8782d4ea6e3e58ea071a9adb9a2e8ddb884a8b" -[[package]] -name = "ndk-sys" -version = "0.5.0+25.2.9519653" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8c196769dd60fd4f363e11d948139556a344e79d451aeb2fa2fd040738ef7691" -dependencies = [ - "jni-sys 0.3.1", -] - [[package]] name = "ndk-sys" version = "0.6.0+11769913" @@ -4743,7 +4640,7 @@ dependencies = [ "oak-undo", "serde_yaml", "smallvec", - "wgpu 29.0.4", + "wgpu", ] [[package]] @@ -4785,13 +4682,14 @@ dependencies = [ name = "oak-core" version = "0.5.0" dependencies = [ + "half", "image", "log", "ocio-rs", "quick-xml 0.41.0", "thiserror 2.0.20", "toml 0.8.23", - "wgpu 25.0.2", + "wgpu", ] [[package]] @@ -4837,7 +4735,7 @@ version = "0.5.0" dependencies = [ "cosmic-text", "libc", - "naga 25.0.1", + "naga", "oak-codec", "oak-core", "oak-node", @@ -4845,7 +4743,7 @@ dependencies = [ "serde", "serde_json", "thiserror 2.0.20", - "wgpu 25.0.2", + "wgpu", ] [[package]] @@ -6989,15 +6887,6 @@ dependencies = [ "lock_api", ] -[[package]] -name = "spirv" -version = "0.3.0+sdk-1.3.268.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eda41003dc44290527a59b13432d4a0379379fa074b70174882adfbdfd917844" -dependencies = [ - "bitflags 2.13.1", -] - [[package]] name = "spirv" version = "0.4.0+sdk-1.4.341.0" @@ -7264,22 +7153,13 @@ version = "0.11.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" -[[package]] -name = "strum" -version = "0.26.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fec0f0aef304996cf250b31b5a10dee7980c85da9d759361292b8bca5a18f06" -dependencies = [ - "strum_macros 0.26.4", -] - [[package]] name = "strum" version = "0.27.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "af23d6f6c1a224baef9d3f61e287d2761385a5b88fdab4eb4c6f11aeb54c4bcf" dependencies = [ - "strum_macros 0.27.2", + "strum_macros", ] [[package]] @@ -7288,19 +7168,6 @@ version = "0.28.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9628de9b8791db39ceda2b119bbe13134770b56c138ec1d3af810d045c04f9bd" -[[package]] -name = "strum_macros" -version = "0.26.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4c6bee85a5a24955dc440386795aa378cd9cf82acd5f764469152d2270e581be" -dependencies = [ - "heck 0.5.0", - "proc-macro2", - "quote", - "rustversion", - "syn 2.0.119", -] - [[package]] name = "strum_macros" version = "0.27.2" @@ -8385,34 +8252,6 @@ version = "0.1.12" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a28ac98ddc8b9274cb41bb4d9d4d5c425b6020c50c46f25559911905610b4a88" -[[package]] -name = "wgpu" -version = "25.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ec8fb398f119472be4d80bc3647339f56eb63b2a331f6a3d16e25d8144197dd9" -dependencies = [ - "arrayvec", - "bitflags 2.13.1", - "cfg_aliases", - "document-features", - "hashbrown 0.15.5", - "js-sys", - "log", - "naga 25.0.1", - "parking_lot", - "portable-atomic", - "profiling", - "raw-window-handle", - "smallvec", - "static_assertions", - "wasm-bindgen", - "wasm-bindgen-futures", - "web-sys", - "wgpu-core 25.0.2", - "wgpu-hal 25.0.2", - "wgpu-types 25.0.0", -] - [[package]] name = "wgpu" version = "29.0.4" @@ -8428,7 +8267,7 @@ dependencies = [ "hashbrown 0.16.1", "js-sys", "log", - "naga 29.0.4", + "naga", "parking_lot", "portable-atomic", "profiling", @@ -8438,40 +8277,9 @@ dependencies = [ "wasm-bindgen", "wasm-bindgen-futures", "web-sys", - "wgpu-core 29.0.4", - "wgpu-hal 29.0.4", - "wgpu-types 29.0.4", -] - -[[package]] -name = "wgpu-core" -version = "25.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f7b882196f8368511d613c6aeec80655160db6646aebddf8328879a88d54e500" -dependencies = [ - "arrayvec", - "bit-set 0.8.0", - "bit-vec 0.8.0", - "bitflags 2.13.1", - "cfg_aliases", - "document-features", - "hashbrown 0.15.5", - "indexmap", - "log", - "naga 25.0.1", - "once_cell", - "parking_lot", - "portable-atomic", - "profiling", - "raw-window-handle", - "rustc-hash 1.1.0", - "smallvec", - "thiserror 2.0.20", - "wgpu-core-deps-apple 25.0.0", - "wgpu-core-deps-emscripten 25.0.0", - "wgpu-core-deps-windows-linux-android 25.0.0", - "wgpu-hal 25.0.2", - "wgpu-types 25.0.0", + "wgpu-core", + "wgpu-hal", + "wgpu-types", ] [[package]] @@ -8490,7 +8298,7 @@ dependencies = [ "hashbrown 0.16.1", "indexmap", "log", - "naga 29.0.4", + "naga", "once_cell", "parking_lot", "portable-atomic", @@ -8499,21 +8307,12 @@ dependencies = [ "rustc-hash 1.1.0", "smallvec", "thiserror 2.0.20", - "wgpu-core-deps-apple 29.0.4", - "wgpu-core-deps-emscripten 29.0.4", - "wgpu-core-deps-windows-linux-android 29.0.4", - "wgpu-hal 29.0.4", + "wgpu-core-deps-apple", + "wgpu-core-deps-emscripten", + "wgpu-core-deps-windows-linux-android", + "wgpu-hal", "wgpu-naga-bridge", - "wgpu-types 29.0.4", -] - -[[package]] -name = "wgpu-core-deps-apple" -version = "25.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cfd488b3239b6b7b185c3b045c39ca6bf8af34467a4c5de4e0b1a564135d093d" -dependencies = [ - "wgpu-hal 25.0.2", + "wgpu-types", ] [[package]] @@ -8522,16 +8321,7 @@ version = "29.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f5e39e26c4c0e07589e67d18546cf79ff45383659fc72fca4dd293358a0347f3" dependencies = [ - "wgpu-hal 29.0.4", -] - -[[package]] -name = "wgpu-core-deps-emscripten" -version = "25.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f09ad7aceb3818e52539acc679f049d3475775586f3f4e311c30165cf2c00445" -dependencies = [ - "wgpu-hal 25.0.2", + "wgpu-hal", ] [[package]] @@ -8540,16 +8330,7 @@ version = "29.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "01e09be551dc939498bdd5f6b2c66e55ab275dad25825267a08605a80fc9f0af" dependencies = [ - "wgpu-hal 29.0.4", -] - -[[package]] -name = "wgpu-core-deps-windows-linux-android" -version = "25.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cba5fb5f7f9c98baa7c889d444f63ace25574833df56f5b817985f641af58e46" -dependencies = [ - "wgpu-hal 25.0.2", + "wgpu-hal", ] [[package]] @@ -8558,54 +8339,7 @@ version = "29.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4e592c1bbef6ad047647ae6e666ebd8cee7a32bb4544d9700ec96cbf73230257" dependencies = [ - "wgpu-hal 29.0.4", -] - -[[package]] -name = "wgpu-hal" -version = "25.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f968767fe4d3d33747bbd1473ccd55bf0f6451f55d733b5597e67b5deab4ad17" -dependencies = [ - "android_system_properties", - "arrayvec", - "ash", - "bit-set 0.8.0", - "bitflags 2.13.1", - "block", - "bytemuck", - "cfg-if", - "cfg_aliases", - "core-graphics-types 0.1.3", - "glow 0.16.0", - "glutin_wgl_sys", - "gpu-alloc", - "gpu-allocator 0.27.0", - "gpu-descriptor", - "hashbrown 0.15.5", - "js-sys", - "khronos-egl", - "libc", - "libloading 0.8.9", - "log", - "metal 0.31.0", - "naga 25.0.1", - "ndk-sys 0.5.0+25.2.9519653", - "objc", - "ordered-float 4.6.0", - "parking_lot", - "portable-atomic", - "profiling", - "range-alloc", - "raw-window-handle", - "renderdoc-sys", - "smallvec", - "thiserror 2.0.20", - "wasm-bindgen", - "web-sys", - "wgpu-types 25.0.0", - "windows 0.58.0", - "windows-core 0.58.0", + "wgpu-hal", ] [[package]] @@ -8623,9 +8357,9 @@ dependencies = [ "bytemuck", "cfg-if", "cfg_aliases", - "glow 0.17.0", + "glow", "glutin_wgl_sys", - "gpu-allocator 0.28.0", + "gpu-allocator", "gpu-descriptor", "hashbrown 0.16.1", "js-sys", @@ -8633,8 +8367,8 @@ dependencies = [ "libc", "libloading 0.8.9", "log", - "naga 29.0.4", - "ndk-sys 0.6.0+11769913", + "naga", + "ndk-sys", "objc2 0.6.4", "objc2-core-foundation", "objc2-foundation 0.3.2", @@ -8656,7 +8390,7 @@ dependencies = [ "wayland-sys", "web-sys", "wgpu-naga-bridge", - "wgpu-types 29.0.4", + "wgpu-types", "windows 0.62.2", "windows-core 0.62.2", "windows-result 0.4.1", @@ -8668,22 +8402,8 @@ version = "29.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "95226013f547544b223281cd16a4fb549aa9dcb562adbda0faae4c73ffbbc161" dependencies = [ - "naga 29.0.4", - "wgpu-types 29.0.4", -] - -[[package]] -name = "wgpu-types" -version = "25.0.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2aa49460c2a8ee8edba3fca54325540d904dd85b2e086ada762767e17d06e8bc" -dependencies = [ - "bitflags 2.13.1", - "bytemuck", - "js-sys", - "log", - "thiserror 2.0.20", - "web-sys", + "naga", + "wgpu-types", ] [[package]] @@ -8759,16 +8479,6 @@ dependencies = [ "windows-targets 0.52.6", ] -[[package]] -name = "windows" -version = "0.58.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dd04d41d93c4992d421894c18c8b43496aa748dd4c081bac0dc93eb0489272b6" -dependencies = [ - "windows-core 0.58.0", - "windows-targets 0.52.6", -] - [[package]] name = "windows" version = "0.61.3" @@ -8837,19 +8547,6 @@ dependencies = [ "windows-targets 0.52.6", ] -[[package]] -name = "windows-core" -version = "0.58.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6ba6d44ec8c2591c134257ce647b7ea6b20335bf6379a27dac5f1641fcf59f99" -dependencies = [ - "windows-implement 0.58.0", - "windows-interface 0.58.0", - "windows-result 0.2.0", - "windows-strings 0.1.0", - "windows-targets 0.52.6", -] - [[package]] name = "windows-core" version = "0.61.2" @@ -8909,17 +8606,6 @@ dependencies = [ "syn 2.0.119", ] -[[package]] -name = "windows-implement" -version = "0.58.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2bbd5b46c938e506ecbce286b6628a02171d56153ba733b6c741fc627ec9579b" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - [[package]] name = "windows-implement" version = "0.60.2" @@ -8942,17 +8628,6 @@ dependencies = [ "syn 2.0.119", ] -[[package]] -name = "windows-interface" -version = "0.58.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "053c4c462dc91d3b1504c6fe5a726dd15e216ba718e84a0e46a88fbe5ded3515" -dependencies = [ - "proc-macro2", - "quote", - "syn 2.0.119", -] - [[package]] name = "windows-interface" version = "0.59.3" @@ -9016,15 +8691,6 @@ dependencies = [ "windows-targets 0.52.6", ] -[[package]] -name = "windows-result" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1d1043d8214f791817bab27572aaa8af63732e11bf84aa21a45a78d6c317ae0e" -dependencies = [ - "windows-targets 0.52.6", -] - [[package]] name = "windows-result" version = "0.3.4" @@ -9043,16 +8709,6 @@ dependencies = [ "windows-link 0.2.1", ] -[[package]] -name = "windows-strings" -version = "0.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4cd9b125c486025df0eabcb585e62173c6c9eddcec5d117d3b6e8c30e2ee4d10" -dependencies = [ - "windows-result 0.2.0", - "windows-targets 0.52.6", -] - [[package]] name = "windows-strings" version = "0.4.2" diff --git a/crates/oak-app/src/oakui/gpu.rs b/crates/oak-app/src/oakui/gpu.rs index ce16ea8bd..1c6e4c86b 100644 --- a/crates/oak-app/src/oakui/gpu.rs +++ b/crates/oak-app/src/oakui/gpu.rs @@ -38,12 +38,111 @@ use std::sync::Mutex; static GPU_CONTEXT: Mutex, std::sync::Arc)>> = Mutex::new(None); -/// Register the window's wgpu device/queue for the 10-bit display path. -/// The app's window builder calls this once per window (last one wins; the -/// renderers all share the same device). +/// The cached CPU-baked display LUT (M2): keyed by the display/color +/// generation so a settings or monitor change rebuilds it, and +/// re-installed whenever the engine context does not have it. +static DISPLAY_LUT: Mutex> = Mutex::new(None); + +/// A key covering every input of the display chain: the displaycolor +/// generation (policy + monitor ICC) and the project's working/output +/// color settings. +fn display_lut_key() -> String { + format!( + "{}|{:?}|{:?}", + super::displaycolor::generation(), + oak_core::color::pipeline_working_space(), + oak_core::color::pipeline_output_spec() + ) +} + +/// Build the working-space → display-device 3D LUT with the exact CPU +/// reference implementation: the output node +/// ([`oak_core::colormath::working_to_display_target`]) followed by the +/// display ICC chain ([`super::displaycolor::apply_f32_rgba`]). This is +/// what makes the GPU present path color-managed: every per-pixel step the +/// CPU path performs runs here once per settings change, on the GPU's +/// behalf, at full precision. +fn build_display_lut() -> oak_core::lut::Lut3d { + let edge = oak_core::lut::Lut3d::DISPLAY_EDGE; + let lo = oak_core::lut::Lut3d::DISPLAY_LO; + let hi = oak_core::lut::Lut3d::DISPLAY_HI; + let n = (edge as usize).pow(3); + let mut samples = vec![0.0f32; n * 4]; + let step = |i: usize, axis: usize| -> f32 { + let t = i as f32 / (edge - 1) as f32; + lo[axis] + (hi[axis] - lo[axis]) * t + }; + for b in 0..edge as usize { + for g in 0..edge as usize { + for r in 0..edge as usize { + let idx = ((b * edge as usize + g) * edge as usize + r) * 4; + samples[idx] = step(r, 0); + samples[idx + 1] = step(g, 1); + samples[idx + 2] = step(b, 2); + samples[idx + 3] = 1.0; + } + } + } + oak_core::colormath::working_to_display_target( + &mut samples, + oak_core::color::pipeline_working_space(), + oak_core::color::pipeline_output_spec(), + ); + super::displaycolor::apply_f32_rgba(&mut samples, n as i64); + let mut data = Vec::with_capacity(n * 3); + for px in samples.chunks_exact(4) { + data.extend_from_slice(&px[..3]); + } + oak_core::lut::Lut3d { edge, lo, hi, data } +} + +/// Install the display LUT on the engine context when missing or stale. +fn ensure_display_lut(ctx: &oak_core::backend::GpuContext) { + let key = display_lut_key(); + let mut cache = DISPLAY_LUT.lock().unwrap_or_else(|e| e.into_inner()); + let fresh = cache.as_ref().is_some_and(|(k, _)| *k == key); + if fresh && ctx.has_display_lut() { + return; + } + let lut = if fresh { + cache + .as_ref() + .map(|(_, l)| l.clone()) + .unwrap_or_else(build_display_lut) + } else { + build_display_lut() + }; + if ctx.set_display_lut(&lut).is_ok() { + *cache = Some((key, lut)); + } +} + +/// Register the window's wgpu device/queue for the 10-bit display path +/// and adopt it into the engine (M2). The engine's render thread then +/// renders on the very device gpui presents with, so finished frames are +/// sampled zero-copy via [`gpui::SurfaceSource::Texture`]. The adoption +/// is a no-op when the engine already created its own device (then +/// [`present_gpu_frame`] reports `None` and the caller stages through the +/// CPU as before). pub fn register_context(device: std::sync::Arc, queue: std::sync::Arc) { if let Ok(mut ctx) = GPU_CONTEXT.lock() { - *ctx = Some((device, queue)); + *ctx = Some((device.clone(), queue.clone())); + } + let adopted = oak_core::backend::GpuContext::adopt(device, queue, oak_core::backend::BackendKind::Auto); + if !oak_core::backend::GpuContext::install_shared(Some(adopted)) { + // The engine context was already used for GPU work before the + // window opened: it cannot be replaced, so present falls back to + // the single staging readback. Log once — this is the only silent + // degradation of the M2 zero-copy path. + static WARNED: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); + if oak_core::backend::GpuContext::shared().is_some_and(|c| !c.is_adopted()) + && !WARNED.swap(true, std::sync::atomic::Ordering::Relaxed) + { + log::error!( + "GPU device adoption refused: the engine created and used a device before the \ + window opened; preview presentation falls back to a staging readback" + ); + } } } @@ -53,6 +152,31 @@ pub fn context_ready() -> bool { GPU_CONTEXT.lock().map(|ctx| ctx.is_some()).unwrap_or(false) } +/// Present an engine-rendered GPU texture (M2): apply the display LUT on +/// the shared device and return the raw `wgpu::Texture` gpui samples. +/// `None` when the texture is CPU-resident, the engine device is not the +/// adopted one, or the LUT pass is unavailable — the caller then falls +/// back to the CPU display path. +pub fn present_gpu_frame( + texture: &oak_core::texture::Texture, +) -> Option> { + let oak_core::texture::Texture::Gpu { token, ctx, .. } = texture else { + return None; + }; + let concrete = ctx + .as_any()? + .downcast_ref::()?; + if !concrete.is_adopted() { + return None; + } + ensure_display_lut(concrete); + let dst = concrete.present_texture(*token).ok()?; + let handle = concrete.texture_handle(dst); + // gpui's `Arc` owns the texture now; release the engine registry entry. + concrete.destroy_texture(dst); + handle +} + /// Upload F32 RGBA samples (tightly packed, `width * height * 4` values) as /// a half-float RGBA16F GPU texture for the 10-bit display path. Returns /// `None` when no context is registered or the samples are malformed — the @@ -117,6 +241,18 @@ pub fn register_display_frame(image_id: usize, width: u32, height: u32, samples: let Some(texture) = upload_rgba16f(width, height, samples) else { return; }; + register_texture(image_id, width, height, texture); +} + +/// Register an already-created `wgpu::Texture` (the M2 zero-copy present +/// result) for `image_id`, so the viewer samples it instead of a CPU +/// upload. The texture must live on the registered window device. +pub fn register_texture( + image_id: usize, + width: u32, + height: u32, + texture: std::sync::Arc, +) { gpui_widgets::viewer::register_gpu_frame( image_id, texture, diff --git a/crates/oak-app/src/oakui/real.rs b/crates/oak-app/src/oakui/real.rs index 0976e6c86..c2f38140f 100644 --- a/crates/oak-app/src/oakui/real.rs +++ b/crates/oak-app/src/oakui/real.rs @@ -708,9 +708,47 @@ fn rendered_to_owned_image(rendered: &super::renderops::RenderedFrame) -> Option super::gpu::register_display_frame(image.id.0, w, h, &samples); Some(Arc::new(image)) } + super::renderops::RenderedFrame::Gpu(texture) => { + // Long-lived caches (full-res fill, thumbnails) must own pixels: + // the explicit CPU readback boundary, then the same display + // chain as the other variants. + let frame = texture.to_frame().ok()?; + let (w, h, mut samples) = samples_from_cpu_frame(&frame)?; + apply_output_node_f32(&mut samples); + super::displaycolor::apply_f32_rgba(&mut samples, (w * h) as i64); + let image = f32_rgba_to_bgra_image(w, h, &samples); + super::gpu::register_display_frame(image.id.0, w, h, &samples); + Some(Arc::new(image)) + } } } +/// Decode a tightly packed/linesize-padded F32 RGBA engine frame into +/// tightly packed samples (the GPU readback path's pixel decode). +fn samples_from_cpu_frame(frame: &oak_core::texture::Frame) -> Option<(u32, u32, Vec)> { + if frame.width <= 0 || frame.height <= 0 { + return None; + } + let (w, h) = (frame.width as usize, frame.height as usize); + let stride = frame.linesize_bytes(); + if frame.data.len() < stride * h { + return None; + } + let mut samples = vec![0.0f32; w * h * 4]; + for y in 0..h { + for (i, px) in frame.data[y * stride..y * stride + w * 16] + .chunks_exact(16) + .enumerate() + { + for c in 0..4 { + samples[(y * w + i) * 4 + c] = + f32::from_ne_bytes([px[c * 4], px[c * 4 + 1], px[c * 4 + 2], px[c * 4 + 3]]); + } + } + } + Some((w as u32, h as u32, samples)) +} + /// The app-side output node for F32 frames (working colorspace → the /// project's output colorspace); pass-through in the legacy working space. fn apply_output_node_f32(samples: &mut [f32]) { @@ -2536,6 +2574,15 @@ impl RealEngine { .collect(); (w, h, bytes) } + super::renderops::RenderedFrame::Gpu(texture) => { + let frame = texture.to_frame().ok()?; + let (w, h, samples) = samples_from_cpu_frame(&frame)?; + let bytes: Vec = samples + .iter() + .map(|v| (v.clamp(0.0, 1.0) * 255.0).round() as u8) + .collect(); + (w, h, bytes) + } }; release_rendered_frame(&rendered); let image = image::RgbaImage::from_raw(width, height, bytes)?; diff --git a/crates/oak-app/src/oakui/renderops.rs b/crates/oak-app/src/oakui/renderops.rs index 0a69deaf3..38bd83e5b 100644 --- a/crates/oak-app/src/oakui/renderops.rs +++ b/crates/oak-app/src/oakui/renderops.rs @@ -544,9 +544,10 @@ pub fn audio_montage(p: &ProjectRef, seq: NodeId, range: TimeRange) -> Vec, }, + /// Thread pipeline: an engine `Texture` — GPU-resident when a device + /// is available (M2 zero-copy present), otherwise an F32 CPU frame + /// from the inline fallback. + Gpu(oak_core::texture::Texture), } impl RenderedFrame { @@ -572,6 +577,7 @@ impl RenderedFrame { match self { RenderedFrame::Shm(f) => f.meta.width, RenderedFrame::CpuF32 { width, .. } => *width, + RenderedFrame::Gpu(texture) => texture.size().0, } } @@ -580,6 +586,7 @@ impl RenderedFrame { match self { RenderedFrame::Shm(f) => f.meta.height, RenderedFrame::CpuF32 { height, .. } => *height, + RenderedFrame::Gpu(texture) => texture.size().1, } } @@ -591,6 +598,7 @@ impl RenderedFrame { match self { RenderedFrame::Shm(f) => f.meta.format, RenderedFrame::CpuF32 { .. } => PIXEL_FORMAT_F32, + RenderedFrame::Gpu(_) => PIXEL_FORMAT_F32, } } @@ -599,6 +607,11 @@ impl RenderedFrame { matches!(self, RenderedFrame::Shm(_)) } + /// True for the thread-pipeline GPU texture variant. + pub fn is_gpu(&self) -> bool { + matches!(self, RenderedFrame::Gpu(_)) + } + /// Build the viewer display image plus the scope samples (M15 S2 /// zero-copy onscreen path). For the shm variant the slot's bytes are /// wrapped into the display buffer — the GPU-upload staging copy, the @@ -645,6 +658,41 @@ impl RenderedFrame { Some((image, scope, None)) } } + RenderedFrame::Gpu(texture) => { + let (w, h) = texture.size(); + if w <= 0 || h <= 0 { + return None; + } + // M2 zero-copy present: when the engine renders on the + // UI's adopted device, the display LUT runs on the GPU and + // the raw texture goes straight to the viewer. A 1×1 + // transparent image keys the GPU texture (the viewer's + // `cpu_image` fallback; the picture itself is the surface). + let presented = super::gpu::present_gpu_frame(texture); + let image = bgra_bytes_to_render_image(1, 1, &[0, 0, 0, 0])?; + if let Some(tex) = presented { + super::gpu::register_texture(image.id.0, w as u32, h as u32, tex); + return Some((image, ScopeData::default(), None)); + } + // Device not shared (e.g. a private engine context): the + // explicit readback boundary, then the CPU display chain. + let frame = texture.to_frame().ok()?; + let (w, h) = (frame.width.max(0) as u32, frame.height.max(0) as u32); + let mut samples = repack_f32_rows( + frame.width, + frame.height, + frame.linesize_bytes() as i32, + &frame.data, + )?; + apply_output_node_f32(&mut samples); + let scope = analyze_f32_rgba(w, h, &samples); + super::displaycolor::apply_f32_rgba(&mut samples, (w * h) as i64); + Some(( + f32_rgba_to_bgra_image(w, h, &samples), + scope, + Some(samples), + )) + } RenderedFrame::CpuF32 { width, height, @@ -849,8 +897,10 @@ fn render_video(params: VideoTicketParams) -> Result { linesize: frame.linesize_bytes() as i32, data: frame.data.clone(), }), + Ok(TicketPayload::Video(texture @ Texture::Gpu { .. })) => { + Ok(RenderedFrame::Gpu(texture.clone())) + } Ok(TicketPayload::ShmFrame(frame)) => Ok(RenderedFrame::Shm(frame.clone())), - Ok(TicketPayload::Video(_)) => Err("render produced a non-CPU frame".to_string()), _ => Err("render produced no video frame".to_string()), } } @@ -1351,6 +1401,62 @@ mod tests { (project, seq, footage) } + /// M2: a GPU-resident thread-pipeline frame presents with zero CPU + /// readback when the engine context is an adopted (shared) device, + /// and falls back to one explicit readback otherwise. + #[test] + fn gpu_frame_to_display_is_zero_copy_on_adopted_context() { + let _media = media_lock(); + let Some(base) = oak_core::backend::gpu_or_skip("the GPU present assertion") else { + return; + }; + let (device, queue) = base.device_queue(); + let adopted = oak_core::backend::GpuContext::adopt( + device, + queue, + oak_core::backend::BackendKind::Auto, + ); + let mut frame = + oak_render::eval::generate_frame(Rational::new(0, 1), (2, 1), oak_core::PixelFormat::F32) + .unwrap(); + for px in frame.data.chunks_exact_mut(16) { + for (c, v) in px.chunks_exact_mut(4).zip([0.25f32, 0.5, 0.75, 1.0]) { + c.copy_from_slice(&v.to_le_bytes()); + } + } + let token = adopted.create_texture(2, 1).unwrap(); + adopted.upload(token, &frame).unwrap(); + let texture = Texture::gpu(adopted.clone(), token, 2, 1, oak_core::PixelFormat::F32); + + oak_core::backend::reset_gpu_transfer_counters(); + let displayed = RenderedFrame::Gpu(texture) + .to_display() + .expect("GPU frame displays"); + assert_eq!( + oak_core::backend::gpu_transfer_counters().1, + 0, + "an adopted-context GPU frame must present without a readback" + ); + assert!( + displayed.2.is_none(), + "the zero-copy path hands a GPU surface, not CPU samples" + ); + + // A private (non-adopted) context cannot be sampled by the UI: + // the display function takes the single explicit readback. + let token = base.create_texture(2, 1).unwrap(); + base.upload(token, &frame).unwrap(); + let private = RenderedFrame::Gpu(Texture::gpu(base.clone(), token, 2, 1, oak_core::PixelFormat::F32)); + oak_core::backend::reset_gpu_transfer_counters(); + let displayed = private.to_display().expect("fallback display"); + assert!(displayed.2.is_some(), "the fallback hands CPU samples"); + assert_eq!( + oak_core::backend::gpu_transfer_counters().1, + 1, + "the fallback is exactly one explicit readback" + ); + } + /// The Chroma Key effect's boolean inputs read as `Boolean(false)` /// through the inspector's parameter path (`effect_params`), so the /// OfxParamsView checkboxes start UNCHECKED (white fill, black border @@ -2108,17 +2214,10 @@ mod tests { oak_core::PixelFormat::F32, ) .expect("graph render"); - let grow; - let gdata; - let goff; - { - let oak_core::texture::Texture::Cpu(ref gf) = &graph_frame else { - panic!("graph render produced a non-CPU frame"); - }; - grow = gf.linesize_bytes(); - gdata = gf.data.clone(); - } - goff = (8 * grow as usize + 8 * 16) as usize; + let gf = graph_frame.to_frame().expect("graph frame readback"); + let grow = gf.linesize_bytes(); + let gdata = gf.data; + let goff = (8 * grow as usize + 8 * 16) as usize; let gr = f32::from_le_bytes(gdata[goff..goff + 4].try_into().unwrap()); assert!( gr > 0.05, @@ -2220,9 +2319,7 @@ mod tests { oak_core::PixelFormat::F32, ) .expect("graph render"); - let oak_core::texture::Texture::Cpu(ref gf) = texture else { - panic!("non-CPU frame"); - }; + let gf = texture.to_frame().expect("graph frame readback"); let stride = gf.linesize_bytes(); let off = (8 * stride as usize + 8 * 16) as usize; ( diff --git a/crates/oak-cli/src/engine.rs b/crates/oak-cli/src/engine.rs index a4abb4050..9144ecf39 100644 --- a/crates/oak-cli/src/engine.rs +++ b/crates/oak-cli/src/engine.rs @@ -673,12 +673,25 @@ pub fn render_frame( data: frame.data.clone(), }) } + Ok(TicketPayload::Video(texture @ oak_core::texture::Texture::Gpu { .. })) => { + // M2: the thread pipeline renders all-GPU; the CLI writes CPU + // pixels, so this is an explicit readback boundary. + let frame = texture + .to_frame() + .map_err(|e| format!("render readback: {e:?}"))?; + Ok(RenderedFrame { + width: frame.width, + height: frame.height, + format: frame.format as i32, + linesize: frame.linesize_bytes() as i32, + data: frame.data, + }) + } Ok(TicketPayload::ShmFrame(frame)) => { let out = shm_to_rendered_frame(frame); m.release_frame(frame); Ok(out) } - Ok(TicketPayload::Video(_)) => Err("render produced a non-CPU frame".to_string()), _ => Err("render produced no video frame".to_string()), } } diff --git a/crates/oak-core/Cargo.toml b/crates/oak-core/Cargo.toml index 3c7d866be..9878ab249 100644 --- a/crates/oak-core/Cargo.toml +++ b/crates/oak-core/Cargo.toml @@ -12,10 +12,13 @@ crate-type = ["rlib"] thiserror = "2.0.20" ocio-rs = { version = "0.2", features = ["bundled"] } # wgpu: portable GPU backend — same major as oak-render's shaderfx/naga -# generation (25); the moved backend/color/texture/frame code is written -# against this API. -wgpu = "25" +# generation (29, matching the gpui_wgpu device the app presents with, so +# the render thread's textures are directly sampleable by the UI). +wgpu = "29" log = "0.4.34" +# half: f16↔f32 conversion for format-aware texture downloads (the M2 +# present target is Rgba16Float; the rest of the pipeline is Rgba32Float). +half = "2" # TOML persistence for the application config (configstore.rs). toml = "0.8" # XML helpers (xmlutils.rs). diff --git a/crates/oak-core/src/backend.rs b/crates/oak-core/src/backend.rs index 19c46547b..b55a6b7e3 100644 --- a/crates/oak-core/src/backend.rs +++ b/crates/oak-core/src/backend.rs @@ -30,11 +30,12 @@ //! surface; when no adapter is available (headless CI, VMs) or the only //! candidates cannot render the pipeline's canonical Rgba32Float target //! (downlevel GL/GLES), it returns `None` and every consumer falls back -//! to the CPU path. GPU tests skip with no adapter. Verified on macOS -//! Metal (wgpu 25.0.2). +//! to the CPU path. GPU tests skip with no adapter unless +//! `OAK_REQUIRE_GPU` is set (CI). Verified on macOS Metal and Linux +//! lavapipe (wgpu 29). use std::collections::HashMap; -use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; use std::sync::{Arc, Mutex, MutexGuard}; use crate::PixelFormat; @@ -212,12 +213,16 @@ impl DisplayBitDepth { } } -/// A GPU-resident texture in the context registry. +/// A GPU-resident texture in the context registry. `Arc` so the present +/// path can hand the very same `wgpu::Texture` to the UI without a copy +/// (M2 zero-copy present): the registry keeps its own reference as long +/// as the engine token lives. #[derive(Clone)] struct GpuTexture { - texture: wgpu::Texture, + texture: Arc, width: u32, height: u32, + format: wgpu::TextureFormat, } fn lock(m: &Mutex) -> MutexGuard<'_, T> { @@ -243,15 +248,35 @@ pub trait GpuContextLike: Send + Sync { dst: u64, processor: Option<&crate::color::ColorProcessor>, ) -> Result<()>; + + /// Concrete-context downcast hook (M2). The present path uses it to + /// reach the LUT/display-pass API and the raw `wgpu::Texture`; + /// trait-only fakes return `None` and callers fall back to CPU + /// delivery. + fn as_any(&self) -> Option<&dyn std::any::Any> { + None + } + + /// The raw texture for a token, for zero-copy presentation. `None` + /// when the context cannot expose it (fakes) or the token is unknown. + fn texture_handle(&self, _token: u64) -> Option> { + None + } } /// The GPU context: one wgpu instance/device/queue for the process, /// plus the texture registry and the blit pipeline. +/// +/// The device may be **adopted** from the host application instead of +/// created here (M2): when the UI and the render thread share one wgpu +/// device, engine textures are directly sampleable by the presenter and +/// the whole present path is zero-copy. `_instance`/`_adapter` are then +/// `None` (the adopter owns them). pub struct GpuContext { // Kept alive for the whole context: the instance must outlive the - // adapter on native backends. - _instance: wgpu::Instance, - _adapter: wgpu::Adapter, + // adapter on native backends. `None` for an adopted context. + _instance: Option, + _adapter: Option, device: wgpu::Device, queue: wgpu::Queue, kind: BackendKind, @@ -264,6 +289,106 @@ pub struct GpuContext { programs: Mutex>>, /// The lazily created 1×1 placeholder texture (unconnected inputs). placeholder: Mutex>, + /// The display LUT texture (M2): the app-installed working-space → + /// display-device transform, applied by [`GpuContext::present_texture`]. + display_lut: Mutex>, + /// The compiled display-LUT passes, keyed by output format. + present: Mutex>, + /// The compiled YUV→RGB pass (M5 decode import dependency). + yuv: Mutex>, + /// Caller-keyed 3D LUT textures (per-node color transforms; M2). + color_luts: Mutex>, + /// Set once this context has created a GPU resource (texture or + /// pipeline). A context that has never touched the GPU can still be + /// replaced by the UI's adopted device (`install_shared`). + used: AtomicBool, +} + +/// The installed display transform: a 3D LUT texture plus its input domain. +struct DisplayLutState { + token: u64, + edge: u32, + lo: [f32; 3], + hi: [f32; 3], +} + +/// Process-wide GPU↔CPU transfer counters (M2 acceptance, modeled on +/// `procpool::main_heap_frame_copies`). `uploads` counts +/// [`GpuContext::upload`] calls (CPU→GPU), `downloads` counts +/// [`GpuContext::download`] calls (GPU→CPU). The GPU graph/present path +/// must not download: readbacks happen only at the three explicit +/// boundaries (CPU OpenFX, export encoder, disk cache) and in tests. +static GPU_UPLOADS: AtomicU64 = AtomicU64::new(0); +static GPU_DOWNLOADS: AtomicU64 = AtomicU64::new(0); + +/// The current (uploads, downloads) GPU transfer counters. +pub fn gpu_transfer_counters() -> (u64, u64) { + ( + GPU_UPLOADS.load(Ordering::Relaxed), + GPU_DOWNLOADS.load(Ordering::Relaxed), + ) +} + +/// Reset the GPU transfer counters (tests). +pub fn reset_gpu_transfer_counters() { + GPU_UPLOADS.store(0, Ordering::Relaxed); + GPU_DOWNLOADS.store(0, Ordering::Relaxed); +} + +/// The process-wide shared-context slot (`shared` / `install_shared`). +struct SharedSlot { + decided: bool, + ctx: Option>, +} + +fn shared_slot() -> &'static Mutex { + static SLOT: std::sync::OnceLock> = std::sync::OnceLock::new(); + SLOT.get_or_init(|| { + Mutex::new(SharedSlot { + decided: false, + ctx: None, + }) + }) +} + +/// Whether GPU-dependent tests must hard-fail when no adapter is +/// available. CI sets `OAK_REQUIRE_GPU=1` on the software-Vulkan runner +/// (lavapipe is present), so a degraded environment fails the suite +/// instead of silently losing the GPU acceptance signal. +pub fn require_gpu_adapter() -> bool { + match std::env::var("OAK_REQUIRE_GPU") { + Ok(v) => !(v == "0" || v.eq_ignore_ascii_case("false")), + Err(_) => false, + } +} + +/// Handle a missing adapter in a GPU acceptance test: panic when +/// `OAK_REQUIRE_GPU` is set, otherwise log the skip (the caller returns). +pub fn skip_or_fail_gpu(what: &str) { + if require_gpu_adapter() { + panic!("no GPU adapter available for {what} (OAK_REQUIRE_GPU is set)"); + } + eprintln!("no GPU adapter; skipping {what}"); +} + +/// A fresh [`GpuContext`] for a test, or `None` when no adapter exists +/// (panics instead when `OAK_REQUIRE_GPU` is set). +pub fn gpu_or_skip(what: &str) -> Option> { + let ctx = GpuContext::create(BackendKind::Auto); + if ctx.is_none() { + skip_or_fail_gpu(what); + } + ctx +} + +/// The process-wide shared context for a test, with the same +/// `OAK_REQUIRE_GPU` policy as [`gpu_or_skip`]. +pub fn shared_gpu_or_skip(what: &str) -> Option> { + let ctx = GpuContext::shared(); + if ctx.is_none() { + skip_or_fail_gpu(what); + } + ctx } // SAFETY check: wgpu Device/Queue/Instance are Send+Sync; the rest is @@ -277,9 +402,9 @@ impl GpuContext { /// path). pub fn create(prefer: BackendKind) -> Option> { for backends in prefer.wgpu_fallbacks() { - let instance = wgpu::Instance::new(&wgpu::InstanceDescriptor { + let instance = wgpu::Instance::new(wgpu::InstanceDescriptor { backends, - ..Default::default() + ..wgpu::InstanceDescriptor::new_without_display_handle() }); let adapter = match pollster_block_on(instance.request_adapter(&wgpu::RequestAdapterOptions { @@ -317,6 +442,7 @@ impl GpuContext { label: Some("oakrender"), required_features, required_limits: wgpu::Limits::default(), + experimental_features: wgpu::ExperimentalFeatures::disabled(), memory_hints: wgpu::MemoryHints::default(), trace: wgpu::Trace::Off, })) { @@ -328,8 +454,8 @@ impl GpuContext { continue; } return Some(Arc::new(Self { - _instance: instance, - _adapter: adapter, + _instance: Some(instance), + _adapter: Some(adapter), device, queue, kind, @@ -339,11 +465,73 @@ impl GpuContext { filterable, programs: Mutex::new(HashMap::new()), placeholder: Mutex::new(None), + display_lut: Mutex::new(None), + present: Mutex::new(Vec::new()), + yuv: Mutex::new(None), + color_luts: Mutex::new(Vec::new()), + used: AtomicBool::new(false), })); } None } + /// Adopt an existing wgpu device/queue (M2): the host UI creates the + /// device (gpui_wgpu) and the engine renders on it, so the textures + /// the render thread produces are directly sampleable by the UI's + /// renderer — the zero-copy present path. `kind` labels the backend + /// (the device does not expose it); `BackendKind::Auto` is acceptable. + pub fn adopt( + device: Arc, + queue: Arc, + kind: BackendKind, + ) -> Arc { + // The adopted device's feature set decides filtering; the render + // pipeline's `Rgba32Float` render target works on every adapter + // whose device made it this far (the adopter only hands over a + // live, validated device). + let filterable = device + .features() + .contains(wgpu::Features::FLOAT32_FILTERABLE); + Arc::new(Self { + _instance: None, + _adapter: None, + device: (*device).clone(), + queue: (*queue).clone(), + kind, + textures: Mutex::new(HashMap::new()), + next_token: AtomicU64::new(1), + blit: Mutex::new(None), + filterable, + programs: Mutex::new(HashMap::new()), + placeholder: Mutex::new(None), + display_lut: Mutex::new(None), + present: Mutex::new(Vec::new()), + yuv: Mutex::new(None), + color_luts: Mutex::new(Vec::new()), + used: AtomicBool::new(false), + }) + } + + /// True when this context wraps a device it did not create (the UI + /// shared its device; presentation is zero-copy on this context). + pub fn is_adopted(&self) -> bool { + self._adapter.is_none() + } + + /// True once the context has created a GPU resource (texture or + /// pipeline). An unused context is replaceable by + /// [`GpuContext::install_shared`]. + pub fn is_used(&self) -> bool { + self.used.load(Ordering::Acquire) + } + + /// The underlying wgpu device/queue. Handing these out lets another + /// context (or the UI) wrap the same device: contexts sharing a device + /// can present each other's textures zero-copy. + pub fn device_queue(&self) -> (Arc, Arc) { + (Arc::new(self.device.clone()), Arc::new(self.queue.clone())) + } + /// The backend actually in use. pub fn kind(&self) -> BackendKind { self.kind @@ -357,34 +545,60 @@ impl GpuContext { /// Create an F32 RGBA texture (the pipeline's canonical format). pub fn create_texture(&self, width: i32, height: i32) -> Result { - if width <= 0 || height <= 0 { + self.create_texture_format( + width, + height, + 1, + wgpu::TextureFormat::Rgba32Float, + wgpu::TextureUsages::TEXTURE_BINDING + | wgpu::TextureUsages::RENDER_ATTACHMENT + | wgpu::TextureUsages::COPY_DST + | wgpu::TextureUsages::COPY_SRC, + ) + } + + /// Create a texture of an arbitrary format/depth (M2): the display + /// LUT is a 3D `R32Float` texture and the YUV→RGB pass reads + /// single-channel plane textures. `depth` > 1 selects `D3`. + pub fn create_texture_format( + &self, + width: i32, + height: i32, + depth: u32, + format: wgpu::TextureFormat, + usage: wgpu::TextureUsages, + ) -> Result { + if width <= 0 || height <= 0 || depth == 0 { return Err(Error::Invalid); } + self.used.store(true, Ordering::Release); let size = wgpu::Extent3d { width: width as u32, height: height as u32, - depth_or_array_layers: 1, + depth_or_array_layers: depth, }; let texture = self.device.create_texture(&wgpu::TextureDescriptor { label: Some("oakrender-texture"), size, mip_level_count: 1, sample_count: 1, - dimension: wgpu::TextureDimension::D2, - format: wgpu::TextureFormat::Rgba32Float, - usage: wgpu::TextureUsages::TEXTURE_BINDING - | wgpu::TextureUsages::RENDER_ATTACHMENT - | wgpu::TextureUsages::COPY_DST - | wgpu::TextureUsages::COPY_SRC, + dimension: if depth > 1 { + wgpu::TextureDimension::D3 + } else { + wgpu::TextureDimension::D2 + }, + format, + usage, view_formats: &[], }); let token = self.next_token.fetch_add(1, Ordering::Relaxed); lock(&self.textures).insert( token, GpuTexture { - texture, + texture: Arc::new(texture), width: width as u32, height: height as u32, + format, }, ); Ok(token) @@ -407,8 +621,676 @@ impl GpuContext { lock(&self.textures).contains_key(&token) } - /// Upload a CPU frame into a texture (F32 RGBA). + /// The raw `wgpu::Texture` behind a token (M2 zero-copy present): the + /// UI hands this very texture to gpui's surface path. The registry + /// keeps its reference, so the caller may drop/destroy its token as + /// soon as the returned `Arc` is stored elsewhere. + pub fn texture_handle(&self, token: u64) -> Option> { + lock(&self.textures).get(&token).map(|t| t.texture.clone()) + } + + /// The texture's format (tests/plane uploads). + pub fn texture_format(&self, token: u64) -> Option { + lock(&self.textures).get(&token).map(|t| t.format) + } + + /// Upload tightly packed raw bytes into a plain-format texture (the + /// YUV→RGB planes). Counted as a CPU→GPU transfer. + pub fn upload_plane(&self, token: u64, data: &[u8]) -> Result<()> { + GPU_UPLOADS.fetch_add(1, Ordering::Relaxed); + let entry = lock(&self.textures) + .get(&token) + .cloned() + .ok_or(Error::NotFound)?; + let w = entry.width; + let h = entry.height; + let bpp = match entry.format { + wgpu::TextureFormat::R8Unorm => 1usize, + wgpu::TextureFormat::R16Unorm | wgpu::TextureFormat::R16Float => 2, + wgpu::TextureFormat::R32Float => 4, + _ => return Err(Error::Invalid), + }; + let row = w as usize * bpp; + if data.len() < row * h as usize { + return Err(Error::Invalid); + } + self.queue.write_texture( + wgpu::TexelCopyTextureInfo { + texture: &entry.texture, + mip_level: 0, + origin: wgpu::Origin3d::ZERO, + aspect: wgpu::TextureAspect::All, + }, + data, + wgpu::TexelCopyBufferLayout { + offset: 0, + bytes_per_row: Some(row as u32), + rows_per_image: None, + }, + wgpu::Extent3d { + width: w, + height: h, + depth_or_array_layers: 1, + }, + ); + Ok(()) + } + + /// Install the display transform LUT (M2). The LUT maps working-space + /// RGB to display-encoded RGB (the output node plus the display ICC + /// chain, baked on the CPU by the app); [`present_texture`] applies it + /// entirely on the GPU. Replaces (and destroys) any previous LUT. + pub fn set_display_lut(&self, lut: &crate::lut::Lut3d) -> Result<()> { + let token = self.upload_lut(lut)?; + let mut slot = lock(&self.display_lut); + if let Some(old) = slot.take() { + self.destroy_texture(old.token); + } + *slot = Some(DisplayLutState { + token, + edge: lut.edge, + lo: lut.lo, + hi: lut.hi, + }); + Ok(()) + } + + /// Upload a 3D LUT as an `Rgba32Float` D3 texture. Counted as a + /// CPU→GPU transfer; callers cache the GPU resource per LUT. + fn upload_lut(&self, lut: &crate::lut::Lut3d) -> Result { + let expected = (lut.edge as usize).pow(3) * 3; + if lut.data.len() != expected { + return Err(Error::Invalid); + } + let token = self.create_texture_format( + lut.edge as i32, + lut.edge as i32, + lut.edge, + wgpu::TextureFormat::Rgba32Float, + wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST, + )?; + // `write_texture` on a 3D texture: one row per (z,y), each row + // `edge` RGBA f32 values, padded to the copy alignment. + let row = lut.edge as usize * 16; + let padded = (row + 255) & !255; + let mut bytes = vec![0u8; padded * lut.edge as usize * lut.edge as usize]; + for z in 0..lut.edge as usize { + for y in 0..lut.edge as usize { + let src = &lut.data[(z * lut.edge as usize + y) * lut.edge as usize * 3..]; + let dst_off = (z * lut.edge as usize + y) * padded; + for x in 0..lut.edge as usize { + bytes[dst_off + x * 16..dst_off + x * 16 + 4] + .copy_from_slice(&src[x * 3].to_le_bytes()); + bytes[dst_off + x * 16 + 4..dst_off + x * 16 + 8] + .copy_from_slice(&src[x * 3 + 1].to_le_bytes()); + bytes[dst_off + x * 16 + 8..dst_off + x * 16 + 12] + .copy_from_slice(&src[x * 3 + 2].to_le_bytes()); + bytes[dst_off + x * 16 + 12..dst_off + x * 16 + 16] + .copy_from_slice(&1.0f32.to_le_bytes()); + } + } + } + let entry = lock(&self.textures) + .get(&token) + .cloned() + .ok_or(Error::NotFound)?; + GPU_UPLOADS.fetch_add(1, Ordering::Relaxed); + self.queue.write_texture( + wgpu::TexelCopyTextureInfo { + texture: &entry.texture, + mip_level: 0, + origin: wgpu::Origin3d::ZERO, + aspect: wgpu::TextureAspect::All, + }, + &bytes, + wgpu::TexelCopyBufferLayout { + offset: 0, + bytes_per_row: Some(padded as u32), + rows_per_image: Some(lut.edge), + }, + wgpu::Extent3d { + width: lut.edge, + height: lut.edge, + depth_or_array_layers: lut.edge, + }, + ); + Ok(token) + } + + /// Whether a display LUT is installed. + pub fn has_display_lut(&self) -> bool { + lock(&self.display_lut).is_some() + } + + /// Apply the installed display LUT to `src` (a working-space + /// `Rgba32Float` texture) and return the resulting `Rgba16Float` + /// display texture token. GPU→GPU: no upload, no download. The caller + /// retrieves the raw handle with [`GpuContext::texture_handle`]. + pub fn present_texture(&self, src: u64) -> Result { + let lut = { + let slot = lock(&self.display_lut); + let Some(lut) = slot.as_ref() else { + return Err(Error::Failed("no display LUT installed".into())); + }; + (lut.token, lut.edge, lut.lo, lut.hi) + }; + self.apply_lut_to( + src, + lut.0, + lut.1, + lut.2, + lut.3, + wgpu::TextureFormat::Rgba16Float, + ) + } + + /// Apply a caller-keyed 3D LUT to `src` and return a new + /// `Rgba32Float` texture token — the GPU color transform for graph + /// nodes (`ColorTransformJob`, M2). The LUT texture is uploaded once + /// per key (the caller passes a stable processor cache id); applying + /// it is GPU→GPU. + pub fn apply_color_lut( + &self, + src: u64, + key: &str, + lut: &crate::lut::Lut3d, + ) -> Result { + let token = { + let mut cache = lock(&self.color_luts); + if let Some((_, token)) = cache.iter().find(|(k, _)| k == key) { + *token + } else { + let token = self.upload_lut(lut)?; + if cache.len() >= 8 { + let (_, old) = cache.remove(0); + self.destroy_texture(old); + } + cache.push((key.to_string(), token)); + token + } + }; + self.apply_lut_to( + src, + token, + lut.edge, + lut.lo, + lut.hi, + wgpu::TextureFormat::Rgba32Float, + ) + } + + /// The shared LUT pass: sample the D3 LUT with manual trilinear + /// interpolation into a new texture of `dst_format`. + fn apply_lut_to( + &self, + src: u64, + lut_token: u64, + edge: u32, + lo: [f32; 3], + hi: [f32; 3], + dst_format: wgpu::TextureFormat, + ) -> Result { + let src_tex = lock(&self.textures) + .get(&src) + .cloned() + .ok_or(Error::NotFound)?; + let lut_tex = lock(&self.textures) + .get(&lut_token) + .cloned() + .ok_or(Error::NotFound)?; + let dst = self.create_texture_format( + src_tex.width as i32, + src_tex.height as i32, + 1, + dst_format, + wgpu::TextureUsages::TEXTURE_BINDING + | wgpu::TextureUsages::RENDER_ATTACHMENT + | wgpu::TextureUsages::COPY_SRC, + )?; + let dst_tex = lock(&self.textures) + .get(&dst) + .cloned() + .ok_or(Error::NotFound)?; + + let pipeline = self.present_pipeline(dst_format)?; + let params: [f32; 12] = [ + edge as f32, + 0.0, + 0.0, + 0.0, + lo[0], + lo[1], + lo[2], + 0.0, + hi[0], + hi[1], + hi[2], + 0.0, + ]; + let uniform = self.device.create_buffer(&wgpu::BufferDescriptor { + label: Some("oakrender-present-params"), + size: 48, + usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST, + mapped_at_creation: false, + }); + self.queue.write_buffer(&uniform, 0, &f32_uniform_bytes(¶ms)); + let src_view = src_tex + .texture + .create_view(&wgpu::TextureViewDescriptor::default()); + let lut_view = lut_tex + .texture + .create_view(&wgpu::TextureViewDescriptor::default()); + let dst_view = dst_tex + .texture + .create_view(&wgpu::TextureViewDescriptor::default()); + let bind_group = self.device.create_bind_group(&wgpu::BindGroupDescriptor { + label: Some("oakrender-present-bg"), + layout: &pipeline.layout, + entries: &[ + wgpu::BindGroupEntry { + binding: 0, + resource: wgpu::BindingResource::TextureView(&src_view), + }, + wgpu::BindGroupEntry { + binding: 1, + resource: wgpu::BindingResource::TextureView(&lut_view), + }, + wgpu::BindGroupEntry { + binding: 2, + resource: uniform.as_entire_binding(), + }, + ], + }); + let mut encoder = self + .device + .create_command_encoder(&wgpu::CommandEncoderDescriptor { + label: Some("oakrender-present"), + }); + { + let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor { + label: Some("oakrender-present-pass"), + color_attachments: &[Some(wgpu::RenderPassColorAttachment { + view: &dst_view, + depth_slice: None, + resolve_target: None, + ops: wgpu::Operations { + load: wgpu::LoadOp::Clear(wgpu::Color::TRANSPARENT), + store: wgpu::StoreOp::Store, + }, + })], + depth_stencil_attachment: None, + timestamp_writes: None, + occlusion_query_set: None, + multiview_mask: None, + }); + pass.set_pipeline(&pipeline.pipeline); + pass.set_bind_group(0, &bind_group, &[]); + pass.draw(0..3, 0..1); + } + self.queue.submit(Some(encoder.finish())); + Ok(dst) + } + + /// The LUT pass (manual trilinear, no float-filtering feature + /// required), cached per output format. + fn present_pipeline(&self, format: wgpu::TextureFormat) -> Result { + let mut cache = lock(&self.present); + if let Some((_, p)) = cache.iter().find(|(f, _)| *f == format) { + return Ok(p.clone()); + } + let layout = self + .device + .create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor { + label: Some("oakrender-present-layout"), + entries: &[ + wgpu::BindGroupLayoutEntry { + binding: 0, + visibility: wgpu::ShaderStages::FRAGMENT, + ty: wgpu::BindingType::Texture { + sample_type: wgpu::TextureSampleType::Float { filterable: false }, + view_dimension: wgpu::TextureViewDimension::D2, + multisampled: false, + }, + count: None, + }, + wgpu::BindGroupLayoutEntry { + binding: 1, + visibility: wgpu::ShaderStages::FRAGMENT, + ty: wgpu::BindingType::Texture { + sample_type: wgpu::TextureSampleType::Float { filterable: false }, + view_dimension: wgpu::TextureViewDimension::D3, + multisampled: false, + }, + count: None, + }, + wgpu::BindGroupLayoutEntry { + binding: 2, + visibility: wgpu::ShaderStages::FRAGMENT, + ty: wgpu::BindingType::Buffer { + ty: wgpu::BufferBindingType::Uniform, + has_dynamic_offset: false, + min_binding_size: None, + }, + count: None, + }, + ], + }); + let vs = self + .device + .create_shader_module(wgpu::ShaderModuleDescriptor { + label: Some("oakrender-present-vs"), + source: wgpu::ShaderSource::Wgsl(std::borrow::Cow::Borrowed(EFFECT_VS_WGSL)), + }); + let fs = self + .device + .create_shader_module(wgpu::ShaderModuleDescriptor { + label: Some("oakrender-present-fs"), + source: wgpu::ShaderSource::Wgsl(std::borrow::Cow::Borrowed(PRESENT_WGSL)), + }); + let pipeline_layout = self + .device + .create_pipeline_layout(&wgpu::PipelineLayoutDescriptor { + label: Some("oakrender-present-pipeline-layout"), + bind_group_layouts: &[Some(&layout)], + immediate_size: 0, + }); + let scope = self.device.push_error_scope(wgpu::ErrorFilter::Validation); + let pipeline = self + .device + .create_render_pipeline(&wgpu::RenderPipelineDescriptor { + label: Some("oakrender-present"), + layout: Some(&pipeline_layout), + vertex: wgpu::VertexState { + module: &vs, + entry_point: Some("vs_main"), + compilation_options: Default::default(), + buffers: &[], + }, + primitive: wgpu::PrimitiveState::default(), + depth_stencil: None, + multisample: wgpu::MultisampleState::default(), + fragment: Some(wgpu::FragmentState { + module: &fs, + entry_point: Some("main"), + compilation_options: Default::default(), + targets: &[Some(wgpu::ColorTargetState { + format, + blend: None, + write_mask: wgpu::ColorWrites::ALL, + })], + }), + multiview_mask: None, + cache: None, + }); + if let Some(err) = pollster_block_on(scope.pop()) { + return Err(Error::Failed(format!("present pipeline validation failed: {err}"))); + } + let program = PresentPipeline { + pipeline, + layout, + }; + cache.push((format, program.clone())); + Ok(program) + } + + /// Run the YUV→RGB pass (M5 dependency): three `R16Float` plane + /// textures (values normalized 0..1) → one `Rgba32Float` destination, + /// with the BT.601/709/2020 matrix and full/limited range encoded in + /// `transform`. This is the GPU replacement for the CPU swscale / + /// `colormath::yuv444p16_to_rgb_f32` conversion; the hardware-decode + /// import path (M5) feeds it imported planes. + pub fn run_yuv_to_rgb( + &self, + y: u64, + u: u64, + v: u64, + dst: u64, + transform: &YuvTransform, + ) -> Result<()> { + let y_tex = lock(&self.textures) + .get(&y) + .cloned() + .ok_or(Error::NotFound)?; + let u_tex = lock(&self.textures) + .get(&u) + .cloned() + .ok_or(Error::NotFound)?; + let v_tex = lock(&self.textures) + .get(&v) + .cloned() + .ok_or(Error::NotFound)?; + let dst_tex = lock(&self.textures) + .get(&dst) + .cloned() + .ok_or(Error::NotFound)?; + let pipeline = self.yuv_pipeline()?; + let m = transform.matrix; + let b = transform.offset; + let params: [f32; 16] = [ + m[0][0], + m[0][1], + m[0][2], + b[0], + m[1][0], + m[1][1], + m[1][2], + b[1], + m[2][0], + m[2][1], + m[2][2], + b[2], + u_tex.width as f32 / y_tex.width as f32, + u_tex.height as f32 / y_tex.height as f32, + 0.0, + 0.0, + ]; + let uniform = self.device.create_buffer(&wgpu::BufferDescriptor { + label: Some("oakrender-yuv-params"), + size: 64, + usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST, + mapped_at_creation: false, + }); + self.queue.write_buffer(&uniform, 0, &f32_uniform_bytes(¶ms)); + let y_view = y_tex + .texture + .create_view(&wgpu::TextureViewDescriptor::default()); + let u_view = u_tex + .texture + .create_view(&wgpu::TextureViewDescriptor::default()); + let v_view = v_tex + .texture + .create_view(&wgpu::TextureViewDescriptor::default()); + let dst_view = dst_tex + .texture + .create_view(&wgpu::TextureViewDescriptor::default()); + let bind_group = self.device.create_bind_group(&wgpu::BindGroupDescriptor { + label: Some("oakrender-yuv-bg"), + layout: &pipeline.layout, + entries: &[ + wgpu::BindGroupEntry { + binding: 0, + resource: wgpu::BindingResource::TextureView(&y_view), + }, + wgpu::BindGroupEntry { + binding: 1, + resource: wgpu::BindingResource::TextureView(&u_view), + }, + wgpu::BindGroupEntry { + binding: 2, + resource: wgpu::BindingResource::TextureView(&v_view), + }, + wgpu::BindGroupEntry { + binding: 3, + resource: uniform.as_entire_binding(), + }, + ], + }); + let mut encoder = self + .device + .create_command_encoder(&wgpu::CommandEncoderDescriptor { + label: Some("oakrender-yuv"), + }); + { + let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor { + label: Some("oakrender-yuv-pass"), + color_attachments: &[Some(wgpu::RenderPassColorAttachment { + view: &dst_view, + depth_slice: None, + resolve_target: None, + ops: wgpu::Operations { + load: wgpu::LoadOp::Clear(wgpu::Color::TRANSPARENT), + store: wgpu::StoreOp::Store, + }, + })], + depth_stencil_attachment: None, + timestamp_writes: None, + occlusion_query_set: None, + multiview_mask: None, + }); + pass.set_pipeline(&pipeline.pipeline); + pass.set_bind_group(0, &bind_group, &[]); + pass.draw(0..3, 0..1); + } + self.queue.submit(Some(encoder.finish())); + Ok(()) + } + + /// The YUV→RGB pass pipeline (built once). + fn yuv_pipeline(&self) -> Result { + let mut cache = lock(&self.yuv); + if let Some(p) = cache.as_ref() { + return Ok(p.clone()); + } + let plane = |binding: u32| wgpu::BindGroupLayoutEntry { + binding, + visibility: wgpu::ShaderStages::FRAGMENT, + ty: wgpu::BindingType::Texture { + sample_type: wgpu::TextureSampleType::Float { filterable: false }, + view_dimension: wgpu::TextureViewDimension::D2, + multisampled: false, + }, + count: None, + }; + let layout = self + .device + .create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor { + label: Some("oakrender-yuv-layout"), + entries: &[ + plane(0), + plane(1), + plane(2), + wgpu::BindGroupLayoutEntry { + binding: 3, + visibility: wgpu::ShaderStages::FRAGMENT, + ty: wgpu::BindingType::Buffer { + ty: wgpu::BufferBindingType::Uniform, + has_dynamic_offset: false, + min_binding_size: None, + }, + count: None, + }, + ], + }); + let vs = self + .device + .create_shader_module(wgpu::ShaderModuleDescriptor { + label: Some("oakrender-yuv-vs"), + source: wgpu::ShaderSource::Wgsl(std::borrow::Cow::Borrowed(EFFECT_VS_WGSL)), + }); + let fs = self + .device + .create_shader_module(wgpu::ShaderModuleDescriptor { + label: Some("oakrender-yuv-fs"), + source: wgpu::ShaderSource::Wgsl(std::borrow::Cow::Borrowed(YUV_WGSL)), + }); + let pipeline_layout = self + .device + .create_pipeline_layout(&wgpu::PipelineLayoutDescriptor { + label: Some("oakrender-yuv-pipeline-layout"), + bind_group_layouts: &[Some(&layout)], + immediate_size: 0, + }); + let scope = self.device.push_error_scope(wgpu::ErrorFilter::Validation); + let pipeline = self + .device + .create_render_pipeline(&wgpu::RenderPipelineDescriptor { + label: Some("oakrender-yuv"), + layout: Some(&pipeline_layout), + vertex: wgpu::VertexState { + module: &vs, + entry_point: Some("vs_main"), + compilation_options: Default::default(), + buffers: &[], + }, + primitive: wgpu::PrimitiveState::default(), + depth_stencil: None, + multisample: wgpu::MultisampleState::default(), + fragment: Some(wgpu::FragmentState { + module: &fs, + entry_point: Some("main"), + compilation_options: Default::default(), + targets: &[Some(wgpu::ColorTargetState { + format: wgpu::TextureFormat::Rgba32Float, + blend: None, + write_mask: wgpu::ColorWrites::ALL, + })], + }), + multiview_mask: None, + cache: None, + }); + if let Some(err) = pollster_block_on(scope.pop()) { + return Err(Error::Failed(format!("YUV pipeline validation failed: {err}"))); + } + let program = PresentPipeline { + pipeline, + layout, + }; + *cache = Some(program.clone()); + Ok(program) + } + + /// Clear a texture to transparent black on the GPU (no CPU transfer): + /// the graph compositor's starting accumulator and single-sided + /// transition sides. + pub fn clear_texture(&self, token: u64) -> Result<()> { + let entry = lock(&self.textures) + .get(&token) + .cloned() + .ok_or(Error::NotFound)?; + let view = entry + .texture + .create_view(&wgpu::TextureViewDescriptor::default()); + let mut encoder = self + .device + .create_command_encoder(&wgpu::CommandEncoderDescriptor { + label: Some("oakrender-clear"), + }); + { + let _pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor { + label: Some("oakrender-clear-pass"), + color_attachments: &[Some(wgpu::RenderPassColorAttachment { + view: &view, + depth_slice: None, + resolve_target: None, + ops: wgpu::Operations { + load: wgpu::LoadOp::Clear(wgpu::Color::TRANSPARENT), + store: wgpu::StoreOp::Store, + }, + })], + depth_stencil_attachment: None, + timestamp_writes: None, + occlusion_query_set: None, + multiview_mask: None, + }); + } + self.queue.submit(Some(encoder.finish())); + Ok(()) + } + + /// Upload a CPU frame into a texture (F32 RGBA). Counted as a CPU→GPU + /// transfer ([`gpu_transfer_counters`]). pub fn upload(&self, token: u64, frame: &Frame) -> Result<()> { + GPU_UPLOADS.fetch_add(1, Ordering::Relaxed); if frame.format != PixelFormat::F32 { return Err(Error::Invalid); } @@ -447,16 +1329,31 @@ impl GpuContext { Ok(()) } - /// Download a texture into a CPU frame. + /// Download a texture into a CPU frame. Counted as a GPU→CPU transfer + /// ([`gpu_transfer_counters`]); the playback path must never take it. + /// + /// Format-aware (M2): `Rgba32Float` copies raw, `Rgba16Float` (the + /// present target) converts half→f32. Other formats are rejected — + /// explicit readback boundaries only ever meet these two. pub fn download(&self, token: u64) -> Result { + GPU_DOWNLOADS.fetch_add(1, Ordering::Relaxed); let entry = lock(&self.textures) .get(&token) .cloned() .ok_or(Error::NotFound)?; + let (bpp, convert_half) = match entry.format { + wgpu::TextureFormat::Rgba32Float => (16usize, false), + wgpu::TextureFormat::Rgba16Float => (8usize, true), + other => { + return Err(Error::Failed(format!( + "texture download: unsupported format {other:?}" + ))) + } + }; let w = entry.width as usize; let h = entry.height as usize; - let linesize = w * 4 * 4; // Rgba32Float - // copy_texture_to_buffer requires a 256-byte-aligned row stride. + let linesize = w * bpp; + // copy_texture_to_buffer requires a 256-byte-aligned row stride. let padded = (linesize + 255) & !255; let buffer = self.device.create_buffer(&wgpu::BufferDescriptor { @@ -502,7 +1399,7 @@ impl GpuContext { .map_async(wgpu::MapMode::Read, move |result| { let _ = tx.send(result.is_ok()); }); - let _ = self.device.poll(wgpu::PollType::wait()); + let _ = self.device.poll(wgpu::PollType::wait_indefinitely()); if !rx.recv().unwrap_or(false) { return Err(Error::Failed("texture download map failed".into())); } @@ -516,6 +1413,16 @@ impl GpuContext { drop(mapped); buffer.unmap(); + if convert_half { + let mut f32_data = vec![0u8; w * h * 16]; + for i in 0..w * h * 4 { + let bits = u16::from_le_bytes([data[i * 2], data[i * 2 + 1]]); + let v = half::f16::from_bits(bits).to_f32(); + f32_data[i * 4..i * 4 + 4].copy_from_slice(&v.to_le_bytes()); + } + data = f32_data; + } + let mut frame = Frame::new(); let mut pod = VideoParamsPod::default(); pod.width = w as i32; @@ -577,6 +1484,7 @@ impl GpuContext { label: Some("oakrender-blit-pass"), color_attachments: &[Some(wgpu::RenderPassColorAttachment { view: &dst_view, + depth_slice: None, resolve_target: None, ops: wgpu::Operations { load: wgpu::LoadOp::Clear(wgpu::Color::TRANSPARENT), @@ -586,6 +1494,7 @@ impl GpuContext { depth_stencil_attachment: None, timestamp_writes: None, occlusion_query_set: None, + multiview_mask: None, }); pass.set_pipeline(&pipeline); pass.set_bind_group(0, &bind_group, &[]); @@ -624,8 +1533,8 @@ impl GpuContext { .device .create_pipeline_layout(&wgpu::PipelineLayoutDescriptor { label: Some("oakrender-blit-layout"), - bind_group_layouts: &[&layout], - push_constant_ranges: &[], + bind_group_layouts: &[Some(&layout)], + immediate_size: 0, }); let pipeline = self .device @@ -651,7 +1560,7 @@ impl GpuContext { write_mask: wgpu::ColorWrites::ALL, })], }), - multiview: None, + multiview_mask: None, cache: None, }); Ok(pipeline) @@ -663,11 +1572,54 @@ impl GpuContext { /// follows the user's `GraphicsBackend` config (`OAK_RENDER_BACKEND` /// overrides). `DisplayRenderer::init` adopts it too, so a process /// owns exactly one wgpu device. + /// + /// The host may install a context first ([`GpuContext::install_shared`], + /// M2): the app hands over the gpui device so the render thread and + /// the presenter share it. Once decided (installed or lazily created) + /// the slot is fixed for the process. pub fn shared() -> Option> { - static SHARED: std::sync::OnceLock>> = std::sync::OnceLock::new(); - SHARED - .get_or_init(|| Self::create(BackendKind::from_user_config())) - .clone() + let mut slot = lock(shared_slot()); + if !slot.decided { + slot.ctx = Self::create(BackendKind::from_user_config()); + slot.decided = true; + } + slot.ctx.clone() + } + + /// Install (or clear) the process-wide shared context (M2). Called by + /// the app before the first render with the UI's adopted device, so + /// the pipeline backend renders on the presenter's device. + /// + /// **Timing guard**: an engine-created context that has not touched + /// the GPU yet is *replaced* — an early [`GpuContext::shared`] call + /// (a thumbnail, a task) must not silently cost the zero-copy present + /// path. Once the incumbent has created any texture or pipeline the + /// replacement is refused (`false`); the caller degrades to the + /// single staging readback. Also `false` when an adopted context is + /// already installed. + pub fn install_shared(ctx: Option>) -> bool { + let mut slot = lock(shared_slot()); + if slot.decided { + let replaceable = ctx.is_some() + && slot + .ctx + .as_ref() + .is_some_and(|c| !c.is_adopted() && !c.is_used()); + if !replaceable { + return false; + } + } + slot.ctx = ctx; + slot.decided = true; + true + } + + /// True when the app installed the shared context (as opposed to the + /// engine lazily creating one from user config). Tests/UI use this to + /// tell "the presenter's device" from "a private device". + pub fn shared_is_installed() -> bool { + let slot = lock(shared_slot()); + slot.decided && slot.ctx.is_some() } /// True when the device can linear-sample Rgba32Float textures @@ -713,6 +1665,7 @@ impl GpuContext { has_uniforms: bool, filtering: bool, ) -> Result> { + self.used.store(true, Ordering::Release); if let Some(p) = lock(&self.programs).get(key) { return Ok(p.clone()); } @@ -781,11 +1734,11 @@ impl GpuContext { .device .create_pipeline_layout(&wgpu::PipelineLayoutDescriptor { label: Some("oakrender-fx-pipeline-layout"), - bind_group_layouts: &[&layout], - push_constant_ranges: &[], + bind_group_layouts: &[Some(&layout)], + immediate_size: 0, }); - self.device.push_error_scope(wgpu::ErrorFilter::Validation); + let scope = self.device.push_error_scope(wgpu::ErrorFilter::Validation); let pipeline = self .device .create_render_pipeline(&wgpu::RenderPipelineDescriptor { @@ -810,10 +1763,10 @@ impl GpuContext { write_mask: wgpu::ColorWrites::ALL, })], }), - multiview: None, + multiview_mask: None, cache: None, }); - if let Some(err) = pollster_block_on(self.device.pop_error_scope()) { + if let Some(err) = pollster_block_on(scope.pop()) { return Err(Error::Failed(format!("effect pipeline validation failed: {err}"))); } @@ -926,6 +1879,7 @@ impl GpuContext { label: Some("oakrender-fx-pass"), color_attachments: &[Some(wgpu::RenderPassColorAttachment { view: &dst_view, + depth_slice: None, resolve_target: None, ops: wgpu::Operations { load: wgpu::LoadOp::Clear(wgpu::Color::TRANSPARENT), @@ -935,6 +1889,7 @@ impl GpuContext { depth_stencil_attachment: None, timestamp_writes: None, occlusion_query_set: None, + multiview_mask: None, }); pass.set_pipeline(&program.pipeline); pass.set_bind_group(0, &bind_group, &[]); @@ -958,6 +1913,191 @@ pub struct ShaderProgram { pub filtering: bool, } +/// The compiled display-LUT pass (M2). +#[derive(Clone)] +struct PresentPipeline { + pipeline: wgpu::RenderPipeline, + layout: wgpu::BindGroupLayout, +} + +/// The YUV→RGB matrix/offset for [`GpuContext::run_yuv_to_rgb`], derived +/// from the same (Kr, Kb) coefficients and range expansion as +/// [`crate::colormath::yuv444p16_to_rgb_f32`]. Inputs are R16Unorm +/// plane textures (normalized 0..1); `rgb = M·yuv + b`. +#[derive(Clone, Copy, Debug, PartialEq)] +pub struct YuvTransform { + /// Row-major 3×3 matrix applied to `(y, u, v)`. + pub matrix: [[f32; 3]; 3], + /// Per-channel offset added after the matrix. + pub offset: [f32; 3], +} + +impl YuvTransform { + /// Build the transform for a luma matrix and range. + pub fn from_matrix(matrix: crate::colormath::YuvMatrix, full_range: bool) -> Self { + let (kr, kb) = matrix.kr_kb(); + let kg = 1.0 - kr - kb; + // CPU reference: Y spans 219<<8 for limited, chroma 224<<8, both + // divided per code value; inputs here are code/65535. + let (ay, by, ac, bc) = if full_range { + (1.0, 0.0, 1.0, 32768.0 / 65535.0) + } else { + ( + 65535.0 / 56064.0, + 4096.0 / 56064.0, + 65535.0 / 57344.0, + 32768.0 / 57344.0, + ) + }; + let rv = 2.0 * (1.0 - kr) * ac; + let bu = 2.0 * (1.0 - kb) * ac; + let gu = -(2.0 * kb * (1.0 - kb) / kg) * ac; + let gv = -(2.0 * kr * (1.0 - kr) / kg) * ac; + Self { + matrix: [[ay, 0.0, rv], [ay, gu, gv], [ay, bu, 0.0]], + offset: [ + -by - 2.0 * (1.0 - kr) * bc, + -by + (2.0 * kr * (1.0 - kr) + 2.0 * kb * (1.0 - kb)) / kg * bc, + -by - 2.0 * (1.0 - kb) * bc, + ], + } + } + + /// BT.601, limited range. + pub fn bt601_limited() -> Self { + Self::from_matrix(crate::colormath::YuvMatrix::Bt601, false) + } + + /// BT.601, full range. + pub fn bt601_full() -> Self { + Self::from_matrix(crate::colormath::YuvMatrix::Bt601, true) + } + + /// BT.709, limited range. + pub fn bt709_limited() -> Self { + Self::from_matrix(crate::colormath::YuvMatrix::Bt709, false) + } + + /// BT.709, full range. + pub fn bt709_full() -> Self { + Self::from_matrix(crate::colormath::YuvMatrix::Bt709, true) + } + + /// BT.2020, limited range. + pub fn bt2020_limited() -> Self { + Self::from_matrix(crate::colormath::YuvMatrix::Bt2020, false) + } + + /// BT.2020, full range. + pub fn bt2020_full() -> Self { + Self::from_matrix(crate::colormath::YuvMatrix::Bt2020, true) + } +} + +/// Pack an f32 slice as little-endian bytes for `write_buffer`. +fn f32_uniform_bytes(values: &[f32]) -> Vec { + let mut out = Vec::with_capacity(values.len() * 4); + for v in values { + out.extend_from_slice(&v.to_le_bytes()); + } + out +} + +/// The display pass: manual trilinear 3D-LUT sampling (R32Float, no +/// float-filtering feature needed) of the working-space pixel, alpha +/// passes through. The LUT is the CPU-baked output node + display ICC +/// chain, so the presentation transform runs entirely on the GPU. +const PRESENT_WGSL: &str = r#" +struct Params { + edge: vec4, + lo: vec4, + hi: vec4, +}; +@group(0) @binding(0) var src_tex: texture_2d; +@group(0) @binding(1) var lut_tex: texture_3d; +@group(0) @binding(2) var p: Params; + + fn lut_at(x: i32, y: i32, z: i32) -> vec3 { + return textureLoad(lut_tex, vec3(x, y, z), 0).rgb; +} + +@fragment +fn main(@builtin(position) frag: vec4) -> @location(0) vec4 { + let dims = textureDimensions(src_tex); + let coord = clamp( + vec2(i32(frag.x), i32(frag.y)), + vec2(0, 0), + vec2(i32(dims.x), i32(dims.y)) - vec2(1, 1), + ); + let c = textureLoad(src_tex, coord, 0); + let n = i32(p.edge.x); + let last = f32(n - 1); + let span = p.hi.xyz - p.lo.xyz; + let u = (c.xyz - p.lo.xyz) / max(span, vec3(1e-12)); + let t = clamp(u, vec3(0.0), vec3(1.0)) * last; + let i0 = vec3(floor(t)); + let f = t - floor(t); + let i1 = min(i0 + vec3(1, 1, 1), vec3(n - 1)); + let c000 = lut_at(i0.x, i0.y, i0.z); + let c100 = lut_at(i1.x, i0.y, i0.z); + let c010 = lut_at(i0.x, i1.y, i0.z); + let c110 = lut_at(i1.x, i1.y, i0.z); + let c001 = lut_at(i0.x, i0.y, i1.z); + let c101 = lut_at(i1.x, i0.y, i1.z); + let c011 = lut_at(i0.x, i1.y, i1.z); + let c111 = lut_at(i1.x, i1.y, i1.z); + let c00 = mix(c000, c100, f.x); + let c10 = mix(c010, c110, f.x); + let c01 = mix(c001, c101, f.x); + let c11 = mix(c011, c111, f.x); + let c0 = mix(c00, c10, f.y); + let c1 = mix(c01, c11, f.y); + let out = mix(c0, c1, f.z); + return vec4(out, c.a); +} +"#; + +/// The YUV→RGB pass (M5 dependency): three `R16Float` plane inputs +/// sampled 1:1 from the frame's Y/U/V planes, one `Rgba32Float` output. +const YUV_WGSL: &str = r#" +struct Params { + m0: vec4, + m1: vec4, + m2: vec4, + uv_scale: vec4, +}; +@group(0) @binding(0) var y_tex: texture_2d; +@group(0) @binding(1) var u_tex: texture_2d; +@group(0) @binding(2) var v_tex: texture_2d; +@group(0) @binding(3) var p: Params; + +@fragment +fn main(@builtin(position) frag: vec4) -> @location(0) vec4 { + let yd = textureDimensions(y_tex); + let coord = clamp( + vec2(i32(frag.x), i32(frag.y)), + vec2(0, 0), + vec2(i32(yd.x), i32(yd.y)) - vec2(1, 1), + ); + let yv = textureLoad(y_tex, coord, 0).r; + let ud = textureDimensions(u_tex); + let uc = clamp( + vec2(vec2(coord) * p.uv_scale.xy), + vec2(0, 0), + vec2(i32(ud.x), i32(ud.y)) - vec2(1, 1), + ); + let uv = textureLoad(u_tex, uc, 0).r; + let vv = textureLoad(v_tex, uc, 0).r; + let yuv = vec3(yv, uv, vv); + let rgb = vec3( + dot(p.m0.xyz, yuv) + p.m0.w, + dot(p.m1.xyz, yuv) + p.m1.w, + dot(p.m2.xyz, yuv) + p.m2.w, + ); + return vec4(rgb, 1.0); +} +"#; + /// The fixed vertex stage for effect passes: a fullscreen triangle /// emitting `ove_texcoord`-convention UVs at location 0. UV v=0 is the /// first texture data row (the upload/download row order), so effect @@ -1007,6 +2147,14 @@ impl GpuContextLike for GpuContext { ) -> Result<()> { self.blit(src, dst, processor) } + + fn as_any(&self) -> Option<&dyn std::any::Any> { + Some(self) + } + + fn texture_handle(&self, token: u64) -> Option> { + self.texture_handle(token) + } } /// Plain-copy blit shader: fullscreen triangle, `textureLoad` (no @@ -1176,14 +2324,13 @@ impl DisplayRenderer { }; ctx.upload(token, &frame)?; } - Ok(Texture::Gpu { + Ok(Texture::gpu( + ctx.clone(), token, - backend: ctx.kind(), width, height, - format: PixelFormat::F32, - ctx: ctx.clone(), - }) + PixelFormat::F32, + )) } else { let mut frame = Frame::new(); let mut pod = *params; @@ -1344,6 +2491,11 @@ pub fn frame_from_pixels_for_upload( mod tests { use super::*; + /// Serializes the tests that assert on the process-global GPU + /// transfer counters (and install a display LUT): parallel tests would + /// otherwise see each other's transfers. + static GPU_COUNTER_LOCK: Mutex<()> = Mutex::new(()); + #[test] fn backend_string_roundtrip() { for (s, kind) in [ @@ -1422,14 +2574,55 @@ mod tests { assert!(GpuContext::create(BackendKind::Cpu).is_none()); } + /// The M2 timing guard: an engine-created context that has not touched + /// the GPU is replaced when the UI installs its adopted device, so an + /// early `shared()` cannot silently cost zero-copy present. A context + /// that has created a texture is no longer replaceable. + #[test] + fn shared_slot_replaces_an_unused_engine_context() { + let Some(base) = any_gpu() else { + return; + }; + assert!( + GpuContext::install_shared(Some(base.clone())), + "the slot starts undecided" + ); + assert!(!GpuContext::shared().unwrap().is_adopted()); + + let (device, queue) = base.device_queue(); + let adopted = + GpuContext::adopt(device, queue, BackendKind::Auto); + assert!( + GpuContext::install_shared(Some(adopted.clone())), + "an unused engine context is replaceable by the UI device" + ); + assert!( + GpuContext::shared().unwrap().is_adopted(), + "the adopted context is now shared" + ); + + // Once the context has created GPU resources it must not be + // pulled out from under in-flight work. + let _texture = adopted.create_texture(2, 2).unwrap(); + let (device, queue) = adopted.device_queue(); + let second = + GpuContext::adopt(device, queue, BackendKind::Auto); + assert!( + !GpuContext::install_shared(Some(second)), + "a used context is not replaceable" + ); + assert!(GpuContext::shared().unwrap().is_adopted()); + } + fn any_gpu() -> Option> { - GpuContext::create(BackendKind::Auto) + // `gpu_or_skip` hard-fails when `OAK_REQUIRE_GPU` is set (CI). + gpu_or_skip("a backend GPU test") } #[test] fn gpu_texture_upload_download_roundtrip() { + let _guard = GPU_COUNTER_LOCK.lock().unwrap_or_else(|e| e.into_inner()); let Some(ctx) = any_gpu() else { - eprintln!("no adapter; skipping GPU round-trip"); return; }; let w = 8; @@ -1460,8 +2653,8 @@ mod tests { #[test] fn gpu_blit_copies_pixels() { + let _guard = GPU_COUNTER_LOCK.lock().unwrap_or_else(|e| e.into_inner()); let Some(ctx) = any_gpu() else { - eprintln!("no adapter; skipping blit"); return; }; let w = 8; @@ -1496,8 +2689,295 @@ mod tests { ctx.destroy_texture(dst); } + /// The GPU present pass reproduces the CPU 3D-LUT evaluation (within + /// the Rgba16Float output quantization) and performs no CPU transfer. + #[test] + fn gpu_present_lut_matches_cpu_trilinear() { + let _guard = GPU_COUNTER_LOCK.lock().unwrap_or_else(|e| e.into_inner()); + let Some(ctx) = any_gpu() else { + return; + }; + // A non-trivial LUT (channel mix + gamma-ish curve). + let lut = crate::lut::Lut3d::build(9, [0.0; 3], [1.0; 3], |c| { + [ + (0.5 * c[0] + 0.25 * c[1] + 0.25 * c[2]).powf(0.8), + c[1].powf(1.2), + (0.2 * c[0] + 0.8 * c[2]).powf(0.9), + ] + }); + ctx.set_display_lut(&lut).unwrap(); + assert!(ctx.has_display_lut()); + + let w = 4; + let h = 3; + let src = ctx.create_texture(w, h).unwrap(); + let mut frame = Frame::new(); + let mut pod = VideoParamsPod::default(); + pod.width = w; + pod.height = h; + frame.set_video_params(pod); + frame.allocate(); + let sample = |i: usize| -> [f32; 4] { + [ + (i as f32 * 0.13) % 1.0, + (i as f32 * 0.29) % 1.0, + (i as f32 * 0.47) % 1.0, + 1.0, + ] + }; + for i in 0..(w * h) as usize { + let px = sample(i); + for (c, v) in px.iter().enumerate() { + frame.data[i * 16 + c * 4..i * 16 + c * 4 + 4] + .copy_from_slice(&v.to_le_bytes()); + } + } + ctx.upload(src, &frame).unwrap(); + // Installation and input upload are one-time boundaries; the + // present itself must transfer nothing to/from the CPU. + reset_gpu_transfer_counters(); + let dst = ctx.present_texture(src).unwrap(); + assert_eq!(gpu_transfer_counters(), (0, 0), "present is GPU→GPU"); + let out = ctx.download(dst).unwrap(); + for i in 0..(w * h) as usize { + let px = sample(i); + let expect = lut.eval([px[0], px[1], px[2]]); + let got = [ + f32::from_le_bytes(out.data[i * 16..i * 16 + 4].try_into().unwrap()), + f32::from_le_bytes(out.data[i * 16 + 4..i * 16 + 8].try_into().unwrap()), + f32::from_le_bytes(out.data[i * 16 + 8..i * 16 + 12].try_into().unwrap()), + ]; + for c in 0..3 { + assert!( + (got[c] - expect[c]).abs() < 5e-3, + "px {i} c{c}: {} vs {}", + got[c], + expect[c] + ); + } + } + ctx.destroy_texture(src); + ctx.destroy_texture(dst); + } + + #[test] + fn gpu_present_requires_a_lut() { + let Some(ctx) = any_gpu() else { + return; + }; + assert!(!ctx.has_display_lut()); + let tex = ctx.create_texture(2, 2).unwrap(); + assert!(ctx.present_texture(tex).is_err()); + ctx.destroy_texture(tex); + } + + #[test] + fn gpu_copy_counters_track_cpu_transfers() { + let _guard = GPU_COUNTER_LOCK.lock().unwrap_or_else(|e| e.into_inner()); + let Some(ctx) = any_gpu() else { + return; + }; + reset_gpu_transfer_counters(); + let token = ctx.create_texture(2, 2).unwrap(); + let mut frame = Frame::new(); + let mut pod = VideoParamsPod::default(); + pod.width = 2; + pod.height = 2; + frame.set_video_params(pod); + frame.allocate(); + ctx.upload(token, &frame).unwrap(); + assert_eq!(gpu_transfer_counters(), (1, 0)); + ctx.download(token).unwrap(); + assert_eq!(gpu_transfer_counters(), (1, 1)); + } + + /// The GPU YUV→RGB pass reproduces the CPU reference conversion + /// (`colormath::yuv444p16_to_rgb_f32`) exactly: same coefficients, same + /// limited/full-range expansion, 4:4:4 planes sampled 1:1. + #[test] + fn gpu_yuv_to_rgb_matches_cpu_reference() { + let _guard = GPU_COUNTER_LOCK.lock().unwrap_or_else(|e| e.into_inner()); + let Some(ctx) = any_gpu() else { + return; + }; + let (w, h) = (8usize, 4usize); + let mut y_code_plane = vec![0u8; w * h * 2]; + let mut u_code_plane = vec![0u8; w * h * 2]; + let mut v_code_plane = vec![0u8; w * h * 2]; + let mut y_plane = vec![0u8; w * h * 2]; + let mut u_plane = vec![0u8; w * h * 2]; + let mut v_plane = vec![0u8; w * h * 2]; + for row in 0..h { + for x in 0..w { + let i = (row * w + x) * 2; + let y_code = (4096 + (x * 8191) % 56064) as u16; + let u_code = (32768i32 - 20000 + (row * 4000) as i32) as u16; + let v_code = (32768i32 + (x * 5000) as i32 - 20000) as u16; + for (code_plane, plane, code) in [ + (&mut y_code_plane, &mut y_plane, y_code), + (&mut u_code_plane, &mut u_plane, u_code), + (&mut v_code_plane, &mut v_plane, v_code), + ] { + code_plane[i..i + 2].copy_from_slice(&code.to_le_bytes()); + let norm = code as f32 / 65535.0; + plane[i..i + 2] + .copy_from_slice(&half::f16::from_f32(norm).to_bits().to_le_bytes()); + } + } + } + let y = ctx + .create_texture_format( + w as i32, + h as i32, + 1, + wgpu::TextureFormat::R16Float, + wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST, + ) + .unwrap(); + let u = ctx + .create_texture_format( + w as i32, + h as i32, + 1, + wgpu::TextureFormat::R16Float, + wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST, + ) + .unwrap(); + let v = ctx + .create_texture_format( + w as i32, + h as i32, + 1, + wgpu::TextureFormat::R16Float, + wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST, + ) + .unwrap(); + ctx.upload_plane(y, &y_plane).unwrap(); + ctx.upload_plane(u, &u_plane).unwrap(); + ctx.upload_plane(v, &v_plane).unwrap(); + let dst = ctx.create_texture(w as i32, h as i32).unwrap(); + + let mut expected = vec![0.0f32; w * h * 4]; + let mut compare = |ctx: &GpuContext, + tag: &str, + matrix: crate::colormath::YuvMatrix, + transform: YuvTransform, + full_range: bool| { + crate::colormath::yuv444p16_to_rgb_f32( + &y_code_plane, + w * 2, + &u_code_plane, + w * 2, + &v_code_plane, + w * 2, + w, + h, + matrix, + full_range, + &mut expected, + ); + ctx.run_yuv_to_rgb(y, u, v, dst, &transform).unwrap(); + let out = ctx.download(dst).unwrap(); + for i in 0..w * h { + let got = [ + f32::from_le_bytes(out.data[i * 16..i * 16 + 4].try_into().unwrap()), + f32::from_le_bytes(out.data[i * 16 + 4..i * 16 + 8].try_into().unwrap()), + f32::from_le_bytes(out.data[i * 16 + 8..i * 16 + 12].try_into().unwrap()), + ]; + for (got_c, want) in got.iter().zip(&expected[i * 4..i * 4 + 3]) { + assert!( + (got_c - want).abs() < 4e-3, + "{tag}: px {i}: got {got_c} want {want}" + ); + } + } + }; + use crate::colormath::YuvMatrix; + compare(&ctx, "bt601 limited", YuvMatrix::Bt601, YuvTransform::bt601_limited(), false); + compare(&ctx, "bt601 full", YuvMatrix::Bt601, YuvTransform::bt601_full(), true); + compare(&ctx, "bt709 limited", YuvMatrix::Bt709, YuvTransform::bt709_limited(), false); + compare(&ctx, "bt709 full", YuvMatrix::Bt709, YuvTransform::bt709_full(), true); + compare(&ctx, "bt2020 limited", YuvMatrix::Bt2020, YuvTransform::bt2020_limited(), false); + compare(&ctx, "bt2020 full", YuvMatrix::Bt2020, YuvTransform::bt2020_full(), true); + + ctx.destroy_texture(y); + ctx.destroy_texture(u); + ctx.destroy_texture(v); + ctx.destroy_texture(dst); + } + + /// The generic color LUT pass (graph `ColorTransformJob`): trilinear + /// LUT application into an `Rgba32Float` texture, GPU→GPU. + #[test] + fn gpu_apply_color_lut_matches_cpu() { + let _guard = GPU_COUNTER_LOCK.lock().unwrap_or_else(|e| e.into_inner()); + let Some(ctx) = any_gpu() else { + return; + }; + let lut = crate::lut::Lut3d::build(9, [0.0; 3], [1.0; 3], |c| { + [c[1], c[2], c[0]] + }); + let (w, h) = (4, 2); + let src = ctx.create_texture(w, h).unwrap(); + let mut frame = Frame::new(); + let mut pod = VideoParamsPod::default(); + pod.width = w; + pod.height = h; + frame.set_video_params(pod); + frame.allocate(); + for i in 0..(w * h) as usize { + let px = [ + (i as f32 * 0.11) % 1.0, + (i as f32 * 0.23) % 1.0, + (i as f32 * 0.37) % 1.0, + 1.0, + ]; + for (c, v) in px.iter().enumerate() { + frame.data[i * 16 + c * 4..i * 16 + c * 4 + 4] + .copy_from_slice(&v.to_le_bytes()); + } + } + ctx.upload(src, &frame).unwrap(); + reset_gpu_transfer_counters(); + let dst = ctx.apply_color_lut(src, "test/swap", &lut).unwrap(); + assert_eq!( + gpu_transfer_counters(), + (1, 0), + "the first apply uploads the LUT once and never reads back" + ); + reset_gpu_transfer_counters(); + let dst2 = ctx.apply_color_lut(src, "test/swap", &lut).unwrap(); + assert_eq!(gpu_transfer_counters(), (0, 0), "cached apply is GPU→GPU"); + let out = ctx.download(dst).unwrap(); + let out2 = ctx.download(dst2).unwrap(); + for got in [&out, &out2] { + for i in 0..(w * h) as usize { + let src_px = [ + f32::from_le_bytes(frame.data[i * 16..i * 16 + 4].try_into().unwrap()), + f32::from_le_bytes(frame.data[i * 16 + 4..i * 16 + 8].try_into().unwrap()), + f32::from_le_bytes(frame.data[i * 16 + 8..i * 16 + 12].try_into().unwrap()), + ]; + let want = lut.eval(src_px); + for c in 0..3 { + let g = f32::from_le_bytes( + got.data[i * 16 + c * 4..i * 16 + c * 4 + 4].try_into().unwrap(), + ); + assert!( + (g - want[c]).abs() < 1e-4, + "px {i} c{c}: {g} vs {}", + want[c] + ); + } + } + } + ctx.destroy_texture(src); + ctx.destroy_texture(dst); + ctx.destroy_texture(dst2); + } + #[test] fn gpu_missing_texture_errors() { + let _guard = GPU_COUNTER_LOCK.lock().unwrap_or_else(|e| e.into_inner()); let Some(ctx) = any_gpu() else { return; }; diff --git a/crates/oak-core/src/colormath.rs b/crates/oak-core/src/colormath.rs index 05f647d50..e5b6fc6fb 100644 --- a/crates/oak-core/src/colormath.rs +++ b/crates/oak-core/src/colormath.rs @@ -886,7 +886,7 @@ pub enum YuvMatrix { impl YuvMatrix { /// The (Kr, Kb) luma-coefficient pair of this matrix. - fn kr_kb(self) -> (f32, f32) { + pub(crate) fn kr_kb(self) -> (f32, f32) { match self { YuvMatrix::Bt601 => (0.299, 0.114), YuvMatrix::Bt709 => (0.2126, 0.0722), diff --git a/crates/oak-core/src/lib.rs b/crates/oak-core/src/lib.rs index dc4254bbd..19977c3b3 100644 --- a/crates/oak-core/src/lib.rs +++ b/crates/oak-core/src/lib.rs @@ -73,6 +73,7 @@ pub mod color; pub mod texture; pub mod frame; pub mod backend; +pub mod lut; pub use handle::CHandle; pub use rational::Rational; diff --git a/crates/oak-core/src/lut.rs b/crates/oak-core/src/lut.rs new file mode 100644 index 000000000..054eae871 --- /dev/null +++ b/crates/oak-core/src/lut.rs @@ -0,0 +1,215 @@ +// Oak Video Editor - Non-Linear Video Editor +// Copyright (C) 2026 Oak Team +// +// This program is free software: you can redistribute it and/or modify +// it under the terms of the GNU General Public License as published by +// the Free Software Foundation, either version 3 of the License, or +// (at your option) any later version. +// +// This program is distributed in the hope that it will be useful, +// but WITHOUT ANY WARRANTY; without even the implied warranty of +// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +// GNU General Public License for more details. +// +// You should have received a copy of the GNU General Public License +// along with this program. If not, see . + +//! CPU-built 3D LUTs for GPU color management (M2). +//! +//! The display pipeline's per-pixel math (working space → project output +//! spec → display device via ICC) is evaluated once per settings change +//! with the exact CPU reference implementation +//! ([`crate::colormath`] + [`crate::color::ColorProcessor`]) and baked +//! into a 3D LUT. The GPU present pass then samples it with manual +//! trilinear interpolation, so the pixels that reach the swapchain are +//! color-managed on the GPU without a CPU readback — the display +//! transform runs at full precision against the same code the CPU path +//! has always used. +//! +//! The LUT is stored as tightly packed f32 RGB, red varying fastest, and +//! covers the cube `[lo, hi]³`; samples outside the domain clamp to the +//! boundary. Alpha is never part of the transform. + +/// A 3D RGB lookup table over `[lo, hi]³`, `edge³` samples, red fastest. +#[derive(Clone, Debug, PartialEq)] +pub struct Lut3d { + /// Samples per axis (≥ 2). + pub edge: u32, + /// Domain low corner. + pub lo: [f32; 3], + /// Domain high corner. + pub hi: [f32; 3], + /// `edge³ * 3` tightly packed RGB values. + pub data: Vec, +} + +impl Lut3d { + /// The standard display-transform edge: 65³ samples keep trilinear + /// error far below the 10-bit display quantization step for the + /// analytic output-node/ICC chains. + pub const DISPLAY_EDGE: u32 = 65; + + /// The display LUT's input domain: working-space scene-linear values + /// can exceed 1 (HDR) and dip negative from gamut matrices. The CPU + /// output node clamps after its gamut matrix, so values outside this + /// range are already clipped by the baked transform. + pub const DISPLAY_LO: [f32; 3] = [-0.25, -0.25, -0.25]; + /// See [`Lut3d::DISPLAY_LO`]. + pub const DISPLAY_HI: [f32; 3] = [4.0, 4.0, 4.0]; + + /// Build a LUT by evaluating `f` on the `edge³` grid. + pub fn build [f32; 3]>( + edge: u32, + lo: [f32; 3], + hi: [f32; 3], + mut f: F, + ) -> Self { + let edge = edge.max(2); + let n = edge as usize; + let mut data = Vec::with_capacity(n * n * n * 3); + let step = |i: usize, axis: usize| -> f32 { + let t = i as f32 / (edge - 1) as f32; + lo[axis] + (hi[axis] - lo[axis]) * t + }; + for b in 0..n { + for g in 0..n { + for r in 0..n { + let out = f([step(r, 0), step(g, 1), step(b, 2)]); + data.extend_from_slice(&out); + } + } + } + Self { edge, lo, hi, data } + } + + /// The identity LUT (useful for pass-through/legacy display paths). + pub fn identity(edge: u32) -> Self { + Self::build(edge, [0.0; 3], [1.0; 3], |c| c) + } + + /// Index of the sample `(r, g, b)` in [`Lut3d::data`]. + fn index(&self, r: u32, g: u32, b: u32) -> usize { + (((b * self.edge + g) * self.edge + r) * 3) as usize + } + + /// One sample (panics if the LUT data is malformed). + pub fn sample(&self, r: u32, g: u32, b: u32) -> [f32; 3] { + let i = self.index(r, g, b); + [self.data[i], self.data[i + 1], self.data[i + 2]] + } + + /// CPU trilinear evaluation (the reference for tests; the GPU pass + /// implements the same interpolation in WGSL). + pub fn eval(&self, rgb: [f32; 3]) -> [f32; 3] { + let last = self.edge - 1; + let mut p = [0.0f32; 3]; + for a in 0..3 { + let span = self.hi[a] - self.lo[a]; + let t = if span.abs() > f32::EPSILON { + (rgb[a] - self.lo[a]) / span + } else { + 0.0 + } + .clamp(0.0, 1.0); + p[a] = t * last as f32; + } + let i0 = [ + p[0].floor() as u32, + p[1].floor() as u32, + p[2].floor() as u32, + ]; + let f = [ + p[0] - i0[0] as f32, + p[1] - i0[1] as f32, + p[2] - i0[2] as f32, + ]; + let i1 = [ + (i0[0] + 1).min(last), + (i0[1] + 1).min(last), + (i0[2] + 1).min(last), + ]; + std::array::from_fn(|a| { + let c00 = lerp( + self.sample(i0[0], i0[1], i0[2])[a], + self.sample(i1[0], i0[1], i0[2])[a], + f[0], + ); + let c10 = lerp( + self.sample(i0[0], i1[1], i0[2])[a], + self.sample(i1[0], i1[1], i0[2])[a], + f[0], + ); + let c01 = lerp( + self.sample(i0[0], i0[1], i1[2])[a], + self.sample(i1[0], i0[1], i1[2])[a], + f[0], + ); + let c11 = lerp( + self.sample(i0[0], i1[1], i1[2])[a], + self.sample(i1[0], i1[1], i1[2])[a], + f[0], + ); + let c0 = lerp(c00, c10, f[1]); + let c1 = lerp(c01, c11, f[1]); + lerp(c0, c1, f[2]) + }) + } + + /// The LUT domain as `(lo, hi)`. + pub fn domain(&self) -> ([f32; 3], [f32; 3]) { + (self.lo, self.hi) + } +} + +fn lerp(a: f32, b: f32, t: f32) -> f32 { + a + (b - a) * t +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn identity_lut_interpolates_linearly() { + let lut = Lut3d::identity(3); + assert_eq!(lut.edge, 3); + let c = lut.eval([0.25, 0.5, 0.75]); + assert!((c[0] - 0.25).abs() < 1e-6); + assert!((c[1] - 0.5).abs() < 1e-6); + assert!((c[2] - 0.75).abs() < 1e-6); + } + + #[test] + fn build_covers_the_domain_corners() { + let lut = Lut3d::build(2, [-1.0; 3], [2.0; 3], |c| [c[0] * 2.0, c[1], c[2]]); + assert_eq!(lut.sample(0, 0, 0), [-2.0, -1.0, -1.0]); + assert_eq!(lut.sample(1, 1, 1), [4.0, 2.0, 2.0]); + // Out-of-domain samples clamp to the nearest corner. + assert_eq!(lut.eval([-100.0, -100.0, -100.0]), [-2.0, -1.0, -1.0]); + assert_eq!(lut.eval([100.0, 100.0, 100.0]), [4.0, 2.0, 2.0]); + } + + #[test] + fn trilinear_matches_manual_blend() { + // A linear LUT is reproduced exactly by trilinear interpolation. + let lut = Lut3d::build(5, [0.0; 3], [1.0; 3], |c| { + [c[0] * 0.5 + c[1] * 0.25 + c[2] * 0.25, c[1], c[2]] + }); + for probe in [[0.1, 0.2, 0.3], [0.9, 0.4, 0.7], [0.0, 1.0, 0.5]] { + let out = lut.eval(probe); + let expect = [ + probe[0] * 0.5 + probe[1] * 0.25 + probe[2] * 0.25, + probe[1], + probe[2], + ]; + for a in 0..3 { + assert!( + (out[a] - expect[a]).abs() < 1e-5, + "axis {a}: {} vs {}", + out[a], + expect[a] + ); + } + } + } +} diff --git a/crates/oak-core/src/texture.rs b/crates/oak-core/src/texture.rs index b3c27fd51..31867b012 100644 --- a/crates/oak-core/src/texture.rs +++ b/crates/oak-core/src/texture.rs @@ -171,14 +171,41 @@ impl Frame { } } +/// Shared ownership of a registry texture (M2). A `Texture::Gpu` value can +/// be cloned freely — the table handles, the graph compositor and the +/// present path all do — and the registry entry is destroyed exactly once, +/// when the last lease drops. Without it, dropping any one clone would +/// pull the texture out from under the others. +pub struct GpuLease { + token: u64, + ctx: Arc, +} + +impl GpuLease { + /// A lease for `token` on `ctx`. + pub fn new(ctx: Arc, token: u64) -> Arc { + Arc::new(Self { token, ctx }) + } + + /// The leased token. + pub fn token(&self) -> u64 { + self.token + } +} + +impl Drop for GpuLease { + fn drop(&mut self) { + self.ctx.destroy_texture(self.token); + } +} + /// A texture: either backend-resident (GPU) or a CPU-frame wrapper. -/// `Clone` is safe: the GPU token destroy is idempotent (registry -/// lookup), so two clones both release safely at their own drop. /// -/// GPU textures carry an `Arc` to their [`GpuContext`] (the C++ `TexturePtr` -/// keeps its renderer alive the same way), so a texture value can upload/ -/// download/blit without a separate renderer handle. `Drop` releases the -/// backend token; destroying a token twice is harmless (registry lookup). +/// GPU textures carry an `Arc` to their [`GpuContext`](crate::backend::GpuContext) +/// (the C++ `TexturePtr` keeps its renderer alive the same way), so a +/// texture value can upload/download/blit without a separate renderer +/// handle, plus a [`GpuLease`] that releases the backend token when the +/// last clone goes away. #[derive(Clone)] pub enum Texture { /// Backend GPU texture. @@ -196,6 +223,8 @@ pub enum Texture { /// The context owning the texture (trait object so tests can fake /// the GPU side; `GpuContext` is the only production implementor). ctx: Arc, + /// Shared token ownership (destroyed with the last clone). + lease: Arc, }, /// CPU-frame wrapper (uploaded lazily by the backend). Cpu(Frame), @@ -224,15 +253,28 @@ impl std::fmt::Debug for Texture { } } -impl Drop for Texture { - fn drop(&mut self) { - if let Texture::Gpu { token, ctx, .. } = self { - ctx.destroy_texture(*token); +impl Texture { + /// Wrap a registry texture (M2): builds the shared [`GpuLease`] so + /// clones release the token exactly once. + pub fn gpu( + ctx: Arc, + token: u64, + width: i32, + height: i32, + format: PixelFormat, + ) -> Self { + let lease = GpuLease::new(ctx.clone(), token); + Texture::Gpu { + token, + backend: ctx.kind(), + width, + height, + format, + ctx, + lease, } } -} -impl Texture { /// A dummy/empty texture (C++ `Texture::dummy` semantics): reads as /// transparent black, never uploaded. pub fn dummy() -> Self { diff --git a/crates/oak-render/Cargo.toml b/crates/oak-render/Cargo.toml index 5ec065e22..10015b37d 100644 --- a/crates/oak-render/Cargo.toml +++ b/crates/oak-render/Cargo.toml @@ -17,13 +17,13 @@ oak-codec = { path = "../oak-codec" } oak-node = { path = "../oak-node" } # wgpu: portable GPU backend (Metal/Vulkan/GL/DX12 in one safe API) — the # direct replacement for the C++ liboakgl2/liboakvulkan backend plugins. -# Version 25 (2025 stable line); the only GPU dependency. -wgpu = "25" +# Version 29 matches gpui_wgpu's device (M2 zero-copy present). +wgpu = "29" # naga: GLSL → WGSL translation for the node shaders (the embedded GLSL # stays the single source of truth, matching the C++ Vulkan backend's -# mechanical-conversion approach). Pinned to wgpu 25's naga generation; +# mechanical-conversion approach). Pinned to wgpu 29's naga generation; # only the GLSL frontend and WGSL writer are compiled. -naga = { version = "25", default-features = false, features = ["glsl-in", "wgsl-out"] } +naga = { version = "29", default-features = false, features = ["glsl-in", "wgsl-out"] } # ocio-rs: safe Rust bindings for OpenColorIO v2.5.2 — the ColorProcessor # implementation; OCIO is never rewritten. The `bundled` feature compiles # the vendored OpenColorIO C++ sources (cmake/ninja required); without it diff --git a/crates/oak-render/src/eval.rs b/crates/oak-render/src/eval.rs index 20454e43f..efad7d8b7 100644 --- a/crates/oak-render/src/eval.rs +++ b/crates/oak-render/src/eval.rs @@ -298,11 +298,12 @@ impl RenderEvalHooks { /// C++ process_color_transform: apply the job's OCIO processor to the /// input texture. A CPU frame converts in place (the real OCIO - /// `convert_frame`); a GPU input passes through with a one-time log — - /// the color-managed GPU blit is deferred at the backend - /// (`GpuContext::blit` rejects a processor). An invalid processor - /// passes the input through unchanged (C++ creates processors - /// non-fatally); a non-texture input is `Error::Invalid`. + /// `convert_frame`); a GPU input gets the same transform baked into a + /// 3D LUT and applied by the GPU LUT pass (M2: color management is + /// never skipped), with a readback+convert+re-upload fallback for + /// exotic processors. An invalid processor passes the input through + /// unchanged (C++ creates processors non-fatally); a non-texture input + /// is `Error::Invalid`. fn process_color_transform_job( &mut self, payload: &ColorTransformJobPayload, @@ -328,19 +329,54 @@ impl RenderEvalHooks { payload.color_processor.convert_frame(frame)?; return Ok(tex); } - // GPU leg: deferred at the backend — pass through, logged once - // per processor. - let key = format!( - "colortransform:gpu:{}", - payload.color_processor.cache_id() - ); - if unsupported_warned().insert(key) { - eprintln!( - "color transform on a GPU texture passes through unchanged: \ - color-managed GPU blit deferred at the backend" - ); + // GPU leg: bake the processor into a 3D LUT and run the GPU color + // pass (no readback). Fall back to an explicit readback + CPU + // conversion + re-upload when the processor cannot be baked or the + // texture's context is not a real GPU context. + let (token, ctx, width, height) = match &tex { + Texture::Gpu { + token, + ctx, + width, + height, + .. + } => (*token, ctx.clone(), *width, *height), + Texture::Cpu(_) => unreachable!("handled above"), + }; + let concrete = ctx + .as_any() + .and_then(|a| a.downcast_ref::()); + if let Some(concrete) = concrete { + if let Some(lut) = color_transform_lut(&payload.color_processor) { + if let Ok(dst) = concrete.apply_color_lut( + token, + &payload.color_processor.cache_id(), + &lut, + ) { + return Ok(Texture::gpu( + ctx, + dst, + width, + height, + PixelFormat::F32, + )); + } + } + let mut frame = tex.to_frame()?; + payload.color_processor.convert_frame(&mut frame)?; + let dst = concrete.create_texture(frame.width, frame.height)?; + concrete.upload(dst, &frame)?; + return Ok(Texture::gpu( + ctx, + dst, + frame.width, + frame.height, + PixelFormat::F32, + )); } - Ok(tex) + let mut frame = tex.to_frame()?; + payload.color_processor.convert_frame(&mut frame)?; + Ok(Texture::wrap_frame(frame)) } /// C++ process_frame_generation: fill the destination with a generated @@ -929,14 +965,13 @@ impl RenderEvalHooks { ctx.destroy_texture(*t); } match result { - Ok(()) => Some(Texture::Gpu { - token: dst, - backend: ctx.kind(), - width: size.0.max(1), - height: size.1.max(1), - format: PixelFormat::F32, - ctx: ctx.clone(), - }), + Ok(()) => Some(Texture::gpu( + ctx.clone(), + dst, + size.0.max(1), + size.1.max(1), + PixelFormat::F32, + )), Err(err) => { ctx.destroy_texture(dst); warn(&format!("run failed: {err:#}")); @@ -1021,6 +1056,18 @@ pub fn render_produced_frame( return render_footage_frame(filename, *stream_index, time, (w, h), format); } + // Generated (transparent) frame: prefer a GPU clear so the pipeline + // stays GPU end to end (M2); fall back to the CPU producer. + if format == PixelFormat::F32 { + if let Some(ctx) = oak_core::backend::GpuContext::shared() { + if let Ok(token) = ctx.create_texture(w, h) { + if ctx.clear_texture(token).is_ok() { + return Ok(Texture::gpu(ctx, token, w, h, PixelFormat::F32)); + } + ctx.destroy_texture(token); + } + } + } let frame = generate_frame(time, (w, h), format)?; Ok(Texture::wrap_frame(frame)) } @@ -1423,46 +1470,80 @@ fn main(@builtin(position) frag: vec4) -> @location(0) vec4 { } "#; -/// GPU composite of `frames` into one `(w, h)` frame: bottom (last) to +/// GPU composite of `frames` into one `(w, h)` texture: bottom (last) to /// top (first), alpha-over into a ping-pong accumulator pair. Frames that /// do not match `(w, h)` are skipped (the caller scales at decode time; /// mismatches are defensive). +/// +/// GPU→GPU (M2): textures already on the context are used by token; CPU +/// frames are uploaded into scratch textures (counted, and only when the +/// caller has a CPU frame in the stack). Nothing is read back — the +/// result stays on the GPU. fn composite_tracks_gpu( - ctx: &oak_core::backend::GpuContext, - frames: &[Frame], - size: (i32, i32), -) -> Result { - let (w, h) = size; - if w <= 0 || h <= 0 { - return Err(Error::Invalid); - } - let program = ctx.compile_shader_pass("oak/builtin/alpha-over", COMP_WGSL, 2, false, false)?; - let mut acc = ctx.create_texture(w, h)?; - let mut out = ctx.create_texture(w, h)?; - let src = ctx.create_texture(w, h)?; - - let result = (|| { - // The accumulator starts fully transparent. - let clear = generate_frame(Rational::new(0, 1), (w, h), PixelFormat::F32)?; - ctx.upload(acc, &clear)?; - for frame in frames.iter().rev().filter(|f| f.width == w && f.height == h) { - ctx.upload(src, frame)?; - ctx.run_shader_pass(&program, &[], &[acc, src], out)?; - std::mem::swap(&mut acc, &mut out); - } - ctx.download(acc) - })(); - - ctx.destroy_texture(acc); - ctx.destroy_texture(out); - ctx.destroy_texture(src); - result + ctx: &std::sync::Arc, + frames: &[Texture], + size: (i32, i32), +) -> Result { + let (w, h) = size; + if w <= 0 || h <= 0 { + return Err(Error::Invalid); + } + let program = ctx.compile_shader_pass("oak/builtin/alpha-over", COMP_WGSL, 2, false, false)?; + let acc = ctx.create_texture(w, h)?; + ctx.clear_texture(acc)?; + let mut scratch: Vec = Vec::new(); + let mut current = acc; + let mut owned_current = true; + let result = (|| -> Result { + for frame in frames.iter().rev().filter(|f| f.size() == (w, h)) { + // Prefer the texture's own context when it is this one; a GPU + // texture from another context can only be read back. + let src_token = match frame { + Texture::Gpu { token, .. } if ctx.has_texture(*token) => *token, + Texture::Gpu { .. } => { + let cpu = frame.to_frame()?; + let t = ctx.create_texture(w, h)?; + ctx.upload(t, &cpu)?; + scratch.push(t); + t + } + Texture::Cpu(f) => { + let t = ctx.create_texture(w, h)?; + ctx.upload(t, f)?; + scratch.push(t); + t + } + }; + let out = ctx.create_texture(w, h)?; + ctx.run_shader_pass(&program, &[], &[current, src_token], out)?; + if owned_current { + ctx.destroy_texture(current); + } + current = out; + owned_current = true; + } + Ok(current) + })(); + for t in scratch { + ctx.destroy_texture(t); + } + match result { + Ok(token) => Ok(Texture::gpu( + ctx.clone(), + token, + w, + h, + PixelFormat::F32, + )), + Err(err) => { + if owned_current { + ctx.destroy_texture(current); + } + Err(err) + } + } } -/// CPU composite of `frames` into one `size` frame — the fallback when no -/// GPU device is available (or the pass fails): bottom (last) to top -/// (first) via [`composite_over`]. -/// /// GPU failures are remembered: a device whose wgpu pipeline fails /// validation (e.g. an adapter that advertises ComputePipeline yet lacks /// the required features) fails EVERY frame otherwise — each attempt @@ -1472,45 +1553,58 @@ fn composite_tracks_gpu( static GPU_COMPOSITE_FAILED: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); -fn composite_tracks(frames: Vec, size: (i32, i32)) -> Frame { - let (w, h) = size; - if w <= 0 || h <= 0 { - return Frame::dummy(); - } - if !GPU_COMPOSITE_FAILED.load(std::sync::atomic::Ordering::Relaxed) { - // Single-frame composites (the common preview case) stay on the - // CPU path: the GPU route costs an upload+download round trip per - // frame for no benefit until two or more tracks overlap. - if frames.len() > 1 { - if let Some(ctx) = oak_core::backend::GpuContext::shared() { - match composite_tracks_gpu(&ctx, &frames, (w, h)) { - Ok(frame) => return frame, - Err(err) => { - eprintln!( - "GPU track composite failed, using CPU (and staying there): {err:#}" - ); - GPU_COMPOSITE_FAILED.store(true, std::sync::atomic::Ordering::Relaxed); - } - } - } - } - } - let Ok(mut acc) = generate_frame(Rational::new(0, 1), (w, h), PixelFormat::F32) else { - return Frame::dummy(); - }; - let acc_stride = acc.linesize_bytes() as i32; - for frame in frames.iter().rev().filter(|f| f.width == w && f.height == h) { - composite_over( - &mut acc.data, - acc_stride, - w, - h, - &frame.data, - frame.linesize_bytes() as i32, - 1.0, - ); - } - acc +/// Composite of `frames` into one `size` texture. The GPU path is +/// preferred whenever a context is available (M2: the graph stays on the +/// GPU end to end); the CPU path is the no-adapter fallback and the +/// explicit (counted) readback for GPU frames that must go through a CPU +/// consumer. +/// +/// Compositing always runs — even a single frame passes through the +/// alpha-over over a transparent accumulator, which is the graph's +/// premultiply step (an adjustment sweep's 0.5-opacity result must become +/// 0.25 after its final composite). +fn composite_tracks(frames: Vec, size: (i32, i32)) -> Texture { + let (w, h) = size; + if w <= 0 || h <= 0 { + return Texture::dummy(); + } + if !GPU_COMPOSITE_FAILED.load(std::sync::atomic::Ordering::Relaxed) { + if let Some(ctx) = oak_core::backend::GpuContext::shared() { + match composite_tracks_gpu(&ctx, &frames, (w, h)) { + Ok(texture) => return texture, + Err(err) => { + eprintln!( + "GPU track composite failed, using CPU (and staying there): {err:#}" + ); + GPU_COMPOSITE_FAILED.store(true, std::sync::atomic::Ordering::Relaxed); + } + } + } + } + // CPU fallback: read every GPU frame back (the explicit boundary) and + // composite the CPU stack. + let Ok(mut acc) = generate_frame(Rational::new(0, 1), (w, h), PixelFormat::F32) else { + return Texture::dummy(); + }; + let acc_stride = acc.linesize_bytes() as i32; + for texture in frames.iter() { + let Ok(frame) = texture.to_frame() else { + continue; + }; + if frame.width != w || frame.height != h { + continue; + } + composite_over( + &mut acc.data, + acc_stride, + w, + h, + &frame.data, + frame.linesize_bytes() as i32, + 1.0, + ); + } + Texture::wrap_frame(acc) } /// One video track's contribution to [`render_graph_frame`], resolved @@ -1558,39 +1652,33 @@ fn layer_progress(in_: Rational, out: Rational, time: Rational) -> f64 { } } -/// Evaluate one block at `time` and read its texture back to a CPU -/// frame. Every non-texture outcome — the evaluator produced no texture -/// channel, the handle is null, or the read-back failed — is `Ok(None)` -/// so a caller can fall back instead of failing the whole frame. +/// Evaluate one block at `time` and return its texture (GPU when the +/// graph produced one). Every non-texture outcome — the evaluator +/// produced no texture channel or the handle is null — is `Ok(None)` so a +/// caller can fall back instead of failing the whole frame. fn evaluate_block_frame( - graph: &oak_node::graph::Graph, - traverser: &mut oak_node::traverser::Traverser, - hooks: &mut RenderEvalHooks, - block: oak_node::id::NodeId, - time: Rational, -) -> Result> { - let request = oak_node::traverser::EvalRequest::new(block, time); - let table = traverser.evaluate(graph, &request, hooks).map_err(|e| { - Error::Failed(format!( - "graph evaluation of block {block:?} failed: {e:?}" - )) - })?; - let Some(NodeValue::Texture(handle)) = table.get(oak_node::value::ValueType::Texture) else { - return Ok(None); - }; - if handle.ctx.is_null() { - return Ok(None); - } - let Some(texture) = (unsafe { oak_node::handle::get_checked::(handle) }).cloned() else { - return Ok(None); - }; - match texture.to_frame() { - Ok(frame) => Ok(Some(frame)), - Err(err) => { - eprintln!("graph sequence: texture read-back failed: {err:#}"); - Ok(None) - } - } + graph: &oak_node::graph::Graph, + traverser: &mut oak_node::traverser::Traverser, + hooks: &mut RenderEvalHooks, + block: oak_node::id::NodeId, + time: Rational, +) -> Result> { + let request = oak_node::traverser::EvalRequest::new(block, time); + let table = traverser.evaluate(graph, &request, hooks).map_err(|e| { + Error::Failed(format!( + "graph evaluation of block {block:?} failed: {e:?}" + )) + })?; + let Some(NodeValue::Texture(handle)) = table.get(oak_node::value::ValueType::Texture) else { + return Ok(None); + }; + if handle.ctx.is_null() { + return Ok(None); + } + let Some(texture) = (unsafe { oak_node::handle::get_checked::(handle) }).cloned() else { + return Ok(None); + }; + Ok(Some(texture)) } /// Blend the two sides of a transition block at `time` with the block's @@ -1609,7 +1697,7 @@ fn blend_transition( time: Rational, progress: f64, shader: &str, -) -> Result> { +) -> Result> { use oak_node::nodes::transitions; use oak_node::value::NodeValueRow; @@ -1627,21 +1715,36 @@ fn blend_transition( // A single-sided transition blends against transparent black (a head // fade-in from nothing, a tail fade-out to nothing) — the same - // shaders, with one side generated empty. + // shaders, with one side generated empty. GPU-resident sides blend + // against a GPU-cleared texture; the CPU fallback generates a frame. let size = from .as_ref() .or(to.as_ref()) - .map(|f| (f.width, f.height)) + .map(|f| f.size()) .or(hooks.frame_size) .unwrap_or((1, 1)); - let black = |size: (i32, i32)| generate_frame(time, size, PixelFormat::F32).ok(); + let black = |size: (i32, i32)| -> Option { + if let Some(ctx) = oak_core::backend::GpuContext::shared() { + let token = ctx.create_texture(size.0, size.1).ok()?; + ctx.clear_texture(token).ok()?; + return Some(Texture::gpu( + ctx, + token, + size.0, + size.1, + PixelFormat::F32, + )); + } + generate_frame(time, size, PixelFormat::F32) + .ok() + .map(Texture::wrap_frame) + }; let from = from.or_else(|| black(size)); let to = to.or_else(|| black(size)); - // Both sides present: blend them. The side frames are boxed as CPU - // textures (the shader-job path uploads those into scratch textures - // for the pass and releases them afterwards), so the originals stay - // around for the fallback below. + // Both sides present: blend them. GPU sides stay GPU (the shader-job + // path consumes their tokens directly); CPU sides upload only inside + // the shader job. The originals stay around for the fallback below. let blended = match (from.as_ref(), to.as_ref()) { (Some(from), Some(to)) => { let payload = ShaderJobPayload { @@ -1654,11 +1757,11 @@ fn blend_transition( params: NodeValueRow::from([ ( transitions::TEXTURE_INPUT.to_string(), - texture_value(Texture::wrap_frame(from.clone())), + texture_value(from.clone()), ), ( transitions::BLEND_INPUT.to_string(), - texture_value(Texture::wrap_frame(to.clone())), + texture_value(to.clone()), ), ( transitions::PROGRESS_INPUT.to_string(), @@ -1667,16 +1770,7 @@ fn blend_transition( ]), iterative_input: String::new(), }; - match hooks.process_shader_job(&payload) { - Some(texture) => match texture.to_frame() { - Ok(frame) => Some(frame), - Err(err) => { - eprintln!("graph sequence: transition read-back failed: {err:#}"); - None - } - }, - None => None, - } + hooks.process_shader_job(&payload) } _ => None, }; @@ -1726,18 +1820,19 @@ fn adjustment_chain_head( /// sweep. /// /// `Ok(None)` means the boundary changes nothing (no effect chain, or the -/// chain produced no readable texture); `Ok(Some(frame))` is the chain's -/// output, which replaces the lower layers. +/// chain produced no texture); `Ok(Some(texture))` is the chain's output, +/// which replaces the lower layers. The sweep stays on the GPU when the +/// collected layers are GPU textures (M2). fn flush_adjustment_layer( graph: &mut oak_node::graph::Graph, traverser: &mut oak_node::traverser::Traverser, hooks: &mut RenderEvalHooks, block: oak_node::id::NodeId, - below: &[Frame], + below: &[Texture], size: (i32, i32), time: Rational, progress: f64, -) -> Result> { +) -> Result> { let Some((head, head_input)) = adjustment_chain_head(graph, block) else { return Ok(None); }; @@ -1748,7 +1843,7 @@ fn flush_adjustment_layer( entry.core.set_standard_value( oak_node::nodes::compositesource::TEXTURE_INPUT, -1, - texture_value(Texture::wrap_frame(below)), + texture_value(below), ); } if let Err(err) = graph.connect(source, head, &head_input, -1) { @@ -1777,13 +1872,7 @@ fn flush_adjustment_layer( else { return Ok(None); }; - match texture.to_frame() { - Ok(frame) => Ok(Some(frame)), - Err(err) => { - eprintln!("graph sequence: adjustment layer read-back failed: {err:#}"); - Ok(None) - } - } + Ok(Some(texture)) } /// Render one frame of `viewer` (a sequence) at `time`: evaluate every @@ -2000,7 +2089,7 @@ pub fn render_graph_frame( let mut perf_collect = perf.then(std::time::Instant::now); let mut perf_collect_ms = 0.0f64; // Topmost frame first (see the track walk above). - let mut frames: Vec = Vec::new(); + let mut frames: Vec = Vec::new(); let mut perf_clip_hist: Vec<(&'static str, oak_node::id::NodeId, f64)> = Vec::new(); for step in steps { match step { @@ -2041,12 +2130,7 @@ pub fn render_graph_frame( else { continue; }; - match texture.to_frame() { - Ok(frame) => frames.insert(0, frame), - Err(err) => { - eprintln!("graph sequence: texture read-back failed: {err:#}") - } - } + frames.insert(0, texture); } } TrackRenderStep::Transition { @@ -2118,7 +2202,11 @@ pub fn render_graph_frame( let composite_started = perf.then(std::time::Instant::now); let mut frame = composite_tracks(frames, size); - frame.timestamp = time; + if let Texture::Cpu(cpu) = &mut frame { + // The GPU path carries no timestamp (there is nowhere to put it); + // the 10-bit present path uses the ticket's time, not this field. + cpu.timestamp = time; + } if perf { let composite_ms = composite_started .map(|t| t.elapsed().as_secs_f64() * 1000.0) @@ -2131,7 +2219,7 @@ pub fn render_graph_frame( perf_collect_ms + composite_ms ); } - Ok(Texture::wrap_frame(frame)) + Ok(frame) } /// Render the audio montage over `params.range` (M12 P1): every clip @@ -2592,6 +2680,67 @@ fn unsupported_warned() -> std::sync::MutexGuard<'static, std::collections::Hash .unwrap_or_else(|e| e.into_inner()) } +/// Build (and cache) the 3D LUT for an OCIO color processor (M2): the +/// CPU reference (`convert_f32_rgba`) fills the grid, and the per-pixel +/// transform then runs on the GPU via `GpuContext::apply_color_lut` — +/// color management is never skipped on a GPU texture. Cached by the +/// processor's OCIO cache id. +fn color_transform_lut( + processor: &oak_core::color::ColorProcessor, +) -> Option> { + type Cache = std::sync::Mutex< + std::collections::HashMap>, + >; + static CACHE: std::sync::OnceLock = std::sync::OnceLock::new(); + let key = processor.cache_id(); + let mut cache = CACHE + .get_or_init(|| std::sync::Mutex::new(std::collections::HashMap::new())) + .lock() + .unwrap_or_else(|e| e.into_inner()); + if let Some(lut) = cache.get(&key) { + return Some(lut.clone()); + } + let lut = build_color_transform_lut(processor)?; + if cache.len() >= 16 { + cache.clear(); + } + cache.insert(key, lut.clone()); + Some(lut) +} + +/// Bake a processor into a 3D LUT over the display domain (scene-linear +/// working values; the same range the presentation LUT covers). +fn build_color_transform_lut( + processor: &oak_core::color::ColorProcessor, +) -> Option> { + use oak_core::lut::Lut3d; + let edge = Lut3d::DISPLAY_EDGE; + let (lo, hi) = (Lut3d::DISPLAY_LO, Lut3d::DISPLAY_HI); + let n = (edge as usize).pow(3); + let mut samples = vec![0.0f32; n * 4]; + let step = |i: usize, axis: usize| -> f32 { + let t = i as f32 / (edge - 1) as f32; + lo[axis] + (hi[axis] - lo[axis]) * t + }; + for b in 0..edge as usize { + for g in 0..edge as usize { + for r in 0..edge as usize { + let idx = ((b * edge as usize + g) * edge as usize + r) * 4; + samples[idx] = step(r, 0); + samples[idx + 1] = step(g, 1); + samples[idx + 2] = step(b, 2); + samples[idx + 3] = 1.0; + } + } + } + processor.convert_f32_rgba(&mut samples, n as i64).ok()?; + let mut data = Vec::with_capacity(n * 3); + for px in samples.chunks_exact(4) { + data.extend_from_slice(&px[..3]); + } + Some(std::sync::Arc::new(Lut3d { edge, lo, hi, data })) +} + /// Log an unsupported-effect passthrough once per type id. fn warn_unsupported_once(type_id: &str, reason: &str) { if unsupported_warned().insert(type_id.to_string()) { @@ -2945,14 +3094,13 @@ mod tests { assert_eq!(f.timestamp, Rational::new(3, 1)); assert!(f.data.iter().all(|&b| b == 0)); // GPU destination rejected. - let mut gpu = Texture::Gpu { - token: 0, - backend: oak_core::backend::BackendKind::Cpu, - width: 8, - height: 8, - format: PixelFormat::F32, - ctx: Arc::new(UnusedCtx), - }; + let mut gpu = Texture::gpu( + Arc::new(UnusedCtx), + 0, + 8, + 8, + PixelFormat::F32, + ); assert!(hooks .process_frame_generation(&mut gpu, Rational::new(1, 1)) .is_err()); @@ -2973,9 +3121,9 @@ mod tests { } fn first_pixel(texture: &Texture) -> [f32; 4] { - let Texture::Cpu(frame) = texture else { - unreachable!() - }; + // GPU textures are read back for the assertion (tests may take + // the counted boundary; the playback path never does). + let frame = texture.to_frame().expect("texture readback"); let mut out = [0f32; 4]; for i in 0..4 { out[i] = f32::from_le_bytes(frame.data[i * 4..i * 4 + 4].try_into().unwrap()); @@ -3445,6 +3593,81 @@ mod tests { let _ = std::fs::remove_file(&path); } + /// M2: a ColorTransformJob on a GPU texture bakes the processor into + /// a 3D LUT and runs the GPU color pass — the transform is applied + /// (not passed through) and the result stays GPU-resident. + #[test] + fn resolve_color_transform_job_applies_lut_on_gpu() { + if oak_core::color::set_up_default_config().is_err() { + eprintln!("bundled OCIO missing; skipping"); + return; + } + let Some(ctx) = oak_core::backend::shared_gpu_or_skip("an eval GPU test") else { + return; + }; + // 1D LUT doubling the red channel (linear ramp 0→0, 1→2). + let path = std::env::temp_dir() + .join(format!("oakrender_lut_double_gpu_{}.cube", std::process::id())); + std::fs::write(&path, "LUT_1D_SIZE 2\n0.0 0.0 0.0\n2.0 1.0 1.0\n").unwrap(); + let Some(processor) = oak_core::color::ColorProcessor::create_lut( + path.to_str().unwrap(), + oak_core::color::Direction::Normal, + ) + .filter(|p| p.is_valid()) else { + eprintln!("LUT processor unavailable; skipping"); + let _ = std::fs::remove_file(&path); + return; + }; + + let mut frame = generate_frame(Rational::new(0, 1), (2, 2), PixelFormat::F32).unwrap(); + for px in frame.data.chunks_exact_mut(16) { + for (c, v) in px.chunks_exact_mut(4).zip([0.25f32, 0.25, 0.25, 1.0]) { + c.copy_from_slice(&v.to_le_bytes()); + } + } + let token = ctx.create_texture(2, 2).unwrap(); + ctx.upload(token, &frame).unwrap(); + let input = Texture::gpu(ctx.clone(), token, 2, 2, PixelFormat::F32); + let payload = ColorTransformJobPayload { + color_processor: std::sync::Arc::new(processor), + input: NodeValue::Texture(oak_node::handle::make_owned(input)), + time: Rational::new(0, 1), + }; + let mut table = NodeValueTable::default(); + table.push( + oak_node::value::ValueType::Texture, + NodeValue::Texture(oak_node::handle::make_owned(Job::ColorTransformJob(payload))), + None, + ); + + use oak_node::traverser::RenderHooks; + let mut hooks = RenderEvalHooks::new(); + hooks.resolve( + oak_node::id::NodeId::INVALID, + &NodeValueRow::new(), + &mut table, + ); + + let rows = table.rows(); + let NodeValue::Texture(handle) = &rows[0].1 else { + unreachable!() + }; + let out = unsafe { oak_node::handle::get_checked::(handle) } + .expect("the job box is replaced by the converted texture"); + assert!( + matches!(out, Texture::Gpu { .. }), + "the GPU color transform stays on the GPU" + ); + let px = first_pixel(out); + assert!( + (px[0] - 0.5).abs() < 5e-3, + "red channel doubles through the GPU LUT: {px:?}" + ); + assert!((px[1] - 0.25).abs() < 5e-3, "green unchanged: {px:?}"); + assert_eq!(px[3], 1.0, "alpha preserved"); + let _ = std::fs::remove_file(&path); + } + /// An invalid processor passes the input texture through unchanged /// (C++ creates processors non-fatally). #[test] @@ -3485,24 +3708,19 @@ mod tests { .expect("the job box is replaced by the input texture"); assert_eq!(first_pixel(out), [0.4, 0.3, 0.2, 1.0], "untouched"); } - /// into transparent, then top (first) over it — `out = src*a + - /// dst*(1-a)`, `out_a = a + dst_a*(1-a)` (premultiplied source). #[test] fn composite_tracks_matches_alpha_over_math() { - let Texture::Cpu(top) = &solid_texture(0.5, 0.25, 0.125, 0.5) else { - unreachable!() - }; - let Texture::Cpu(bottom) = &solid_texture(1.0, 1.0, 1.0, 0.75) else { - unreachable!() - }; - let frames = vec![top.clone(), bottom.clone()]; + let frames = vec![ + solid_texture(0.5, 0.25, 0.125, 0.5), + solid_texture(1.0, 1.0, 1.0, 0.75), + ]; let expected = [0.625f32, 0.5, 0.4375, 0.875]; // bottom over transparent: (0.75, 0.75, 0.75, 0.75), then top over: // r = 0.5*0.5 + 0.75*0.5, g = 0.25*0.5 + 0.75*0.5, // b = 0.125*0.5 + 0.75*0.5, a = 0.5 + 0.75*0.5. let out = composite_tracks(frames.clone(), (2, 1)); - let pixel = first_pixel(&Texture::wrap_frame(out)); + let pixel = first_pixel(&out); for (got, want) in pixel.iter().zip(expected) { assert!((got - want).abs() < 1e-4, "CPU composite: expected {want}, got {got}"); } @@ -3510,7 +3728,7 @@ mod tests { // Same math through the GPU pass when a device is available. if let Some(ctx) = oak_core::backend::GpuContext::shared() { let gpu_out = composite_tracks_gpu(&ctx, &frames, (2, 1)).expect("GPU composite"); - let pixel = first_pixel(&Texture::wrap_frame(gpu_out)); + let pixel = first_pixel(&gpu_out); for (got, want) in pixel.iter().zip(expected) { assert!((got - want).abs() < 1e-3, "GPU composite: expected {want}, got {got}"); } @@ -3529,8 +3747,7 @@ mod tests { /// under their (fixed) input-id spelling. #[test] fn gpu_chromakey_keys_green_with_ocio_stub() { - let Some(ctx) = oak_core::backend::GpuContext::shared() else { - eprintln!("no adapter; skipping"); + let Some(ctx) = oak_core::backend::shared_gpu_or_skip("an eval GPU test") else { return; }; // Install the process-wide default config (the C++ @@ -3562,14 +3779,7 @@ mod tests { let src = ctx.create_texture(size.0, size.1).unwrap(); ctx.upload(src, &frame).unwrap(); - let input = Texture::Gpu { - token: src, - backend: ctx.kind(), - width: size.0, - height: size.1, - format: PixelFormat::F32, - ctx: ctx.clone(), - }; + let input = Texture::gpu(ctx.clone(), src, size.0, size.1, PixelFormat::F32); let mut params = NodeValueRow::new(); params.insert("tex_in".into(), NodeValue::Texture(oak_node::handle::make_owned(input))); @@ -3673,8 +3883,7 @@ mod tests { /// run at all. Red base + half-alpha green blend -> (0.5, 1, 0, 1). #[test] fn gpu_merge_alpha_over_binds_base_and_blend() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("no adapter; skipping"); + if oak_core::backend::shared_gpu_or_skip("an eval GPU test").is_none() { return; } let mut inputs = NodeValueRow::new(); @@ -3704,8 +3913,7 @@ mod tests { /// size instead of a 1x1 the composite step would drop. #[test] fn gpu_generator_without_input_renders_at_frame_size() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("no adapter; skipping"); + if oak_core::backend::shared_gpu_or_skip("an eval GPU test").is_none() { return; } let mut inputs = NodeValueRow::new(); @@ -3731,8 +3939,7 @@ mod tests { /// (the generated layer), the corners stay red (the base). #[test] fn gpu_generator_over_base_composites_nested_job() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("no adapter; skipping"); + if oak_core::backend::shared_gpu_or_skip("an eval GPU test").is_none() { return; } let mut inputs = NodeValueRow::new(); @@ -3760,8 +3967,7 @@ mod tests { /// from the source, widening the non-transparent area. #[test] fn gpu_dropshadow_softness_blurs_and_offsets() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("no adapter; skipping"); + if oak_core::backend::shared_gpu_or_skip("an eval GPU test").is_none() { return; } // 16x16 transparent frame with an opaque 4x4 square at (4,4). @@ -3800,8 +4006,7 @@ mod tests { /// translation moves the white pixel from (2, 3) to (5, 3). #[test] fn gpu_transform_translates_pixels() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("no adapter; skipping"); + if oak_core::backend::shared_gpu_or_skip("an eval GPU test").is_none() { return; } // 8x8 black frame with one white pixel at (2, 3). @@ -3827,8 +4032,7 @@ mod tests { /// with no scaling or smearing (the turn is lossless). #[test] fn gpu_transform_rotates_around_the_frame_center() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("no adapter; skipping"); + if oak_core::backend::shared_gpu_or_skip("an eval GPU test").is_none() { return; } // 8x8 black frame with one white pixel at (6, 4) — center+(2, 0). @@ -3861,8 +4065,7 @@ mod tests { /// corner pivot would drag it toward the bottom-right instead). #[test] fn gpu_transform_scales_around_the_frame_center() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("no adapter; skipping"); + if oak_core::backend::shared_gpu_or_skip("an eval GPU test").is_none() { return; } // 8x8 black frame with a white 2x2 block at texels (3..4, 3..4). @@ -3907,8 +4110,7 @@ mod tests { /// translation vacates exactly the left three columns. #[test] fn gpu_transform_off_frame_is_transparent() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("no adapter; skipping"); + if oak_core::backend::shared_gpu_or_skip("an eval GPU test").is_none() { return; } let white = filled_frame((8, 8), [1.0, 1.0, 1.0, 1.0]); @@ -3948,8 +4150,7 @@ mod tests { /// transparent. #[test] fn gpu_shape_rectangle_draws_centered_block() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("no adapter; skipping"); + if oak_core::backend::shared_gpu_or_skip("an eval GPU test").is_none() { return; } let mut inputs = NodeValueRow::new(); @@ -3977,8 +4178,7 @@ mod tests { /// (radius 20 clamps to half the 8px size). #[test] fn gpu_shape_ellipse_and_rounded_rect() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("no adapter; skipping"); + if oak_core::backend::shared_gpu_or_skip("an eval GPU test").is_none() { return; } let base_inputs = || { @@ -4014,8 +4214,7 @@ mod tests { /// dispatch must survive translation). #[test] fn gpu_despill_average_caps_green() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("no adapter; skipping"); + if oak_core::backend::shared_gpu_or_skip("an eval GPU test").is_none() { return; } let mut inputs = NodeValueRow::new(); @@ -4053,8 +4252,7 @@ mod tests { fn bfs_endpoint_sweep_renders_footage_through_position() { use oak_node::nodes::graphendpoints::{GRAPH_INPUT_FEED_INPUT, GRAPH_OUTPUT_INPUT}; - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("no adapter; skipping"); + if oak_core::backend::shared_gpu_or_skip("an eval GPU test").is_none() { return; } diff --git a/crates/oak-render/tests/adjustment_layer.rs b/crates/oak-render/tests/adjustment_layer.rs index 9c0473b6d..03ec2fd63 100644 --- a/crates/oak-render/tests/adjustment_layer.rs +++ b/crates/oak-render/tests/adjustment_layer.rs @@ -195,11 +195,10 @@ fn build_project( } /// The raw CPU frame bytes of a rendered texture. -fn frame_data(texture: &Texture) -> &[u8] { - let Texture::Cpu(frame) = texture else { - panic!("graph render produced a non-CPU texture"); - }; - &frame.data +/// The raw frame bytes of a rendered texture (GPU textures are read back +/// for the assertion; the playback path itself never downloads). +fn frame_data(texture: &Texture) -> Vec { + texture.to_frame().expect("graph frame readback").data } /// Render one 64x64 F32 frame of `seq` at `time` and return its bytes. @@ -207,7 +206,7 @@ fn render_frame(project: &Arc>, seq: NodeId, time: Rational) -> V let texture = oak_render::eval::render_graph_frame(project, seq, time, (64, 64), PixelFormat::F32) .expect("graph render"); - frame_data(&texture).to_vec() + frame_data(&texture) } /// The F32 RGBA channel of a 64x64 frame at `(x, y)`. @@ -229,8 +228,9 @@ fn channel(data: &[u8], x: usize, y: usize, c: usize) -> f32 { /// Skipped (with a note) when no GPU adapter exists. #[test] fn adjustment_layer_affects_lower_tracks_across_clips() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("skipping adjustment_layer_affects_lower_tracks_across_clips: no GPU adapter"); + if oak_core::backend::shared_gpu_or_skip("adjustment_layer_affects_lower_tracks_across_clips") + .is_none() + { return; } let red = clip_path("across_red"); diff --git a/crates/oak-render/tests/graph_render.rs b/crates/oak-render/tests/graph_render.rs index 31a701e4c..5bf341b7a 100644 --- a/crates/oak-render/tests/graph_render.rs +++ b/crates/oak-render/tests/graph_render.rs @@ -131,11 +131,10 @@ fn build_project(clips: &[(&str, Rational, Rational)]) -> (Arc>, } /// The raw CPU frame bytes of a rendered texture. -fn frame_data(texture: &Texture) -> &[u8] { - let Texture::Cpu(frame) = texture else { - panic!("graph render produced a non-CPU texture"); - }; - &frame.data +/// The raw frame bytes of a rendered texture (GPU textures are read back +/// for the assertion; the playback path itself never downloads). +fn frame_data(texture: &Texture) -> Vec { + texture.to_frame().expect("graph frame readback").data } /// Two clips on two tracks, non-overlapping in time: at each request time @@ -228,10 +227,10 @@ fn graph_sequence_stacks_highest_track_on_top() { .expect("stacked render"); let data = frame_data(&tex); assert!( - channel(data, 8, 8, 2) > 0.5 && channel(data, 8, 8, 0) < 0.4, + channel(&data, 8, 8, 2) > 0.5 && channel(&data, 8, 8, 0) < 0.4, "V2's blue covers V1's red (r={}, b={})", - channel(data, 8, 8, 0), - channel(data, 8, 8, 2) + channel(&data, 8, 8, 0), + channel(&data, 8, 8, 2) ); // Distinguishability guard: solo, the V1 clip really is red (the two @@ -241,10 +240,10 @@ fn graph_sequence_stacks_highest_track_on_top() { .expect("solo V1 render"); let solo_data = frame_data(&solo_tex); assert!( - channel(solo_data, 8, 8, 0) > 0.5 && channel(solo_data, 8, 8, 2) < 0.4, + channel(&solo_data, 8, 8, 0) > 0.5 && channel(&solo_data, 8, 8, 2) < 0.4, "solo V1 is red (r={}, b={})", - channel(solo_data, 8, 8, 0), - channel(solo_data, 8, 8, 2) + channel(&solo_data, 8, 8, 0), + channel(&solo_data, 8, 8, 2) ); let _ = std::fs::remove_file(&red); @@ -366,8 +365,7 @@ fn channel(data: &[u8], x: usize, y: usize, c: usize) -> f32 { /// when no GPU adapter exists. #[test] fn shader_job_opacity_halves_pixels() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("skipping shader_job_opacity_halves_pixels: no GPU adapter"); + if oak_core::backend::shared_gpu_or_skip("shader_job_opacity_halves_pixels").is_none() { return; } let path = clip_path("opacity_job"); @@ -452,8 +450,7 @@ fn shader_job_opacity_halves_pixels() { /// boundary pixel on the right half. Skipped when no GPU adapter exists. #[test] fn shader_job_blur_smooths_edge() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("skipping shader_job_blur_smooths_edge: no GPU adapter"); + if oak_core::backend::shared_gpu_or_skip("shader_job_blur_smooths_edge").is_none() { return; } let path = clip_path("blur_job"); @@ -554,8 +551,8 @@ fn shader_job_blur_smooths_edge() { /// Skipped when no GPU adapter or OCIO config exists. #[test] fn chromakey_job_keys_green_with_ociobased_stub() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("skipping chromakey_job_keys_green_with_ociobased_stub: no GPU adapter"); + if oak_core::backend::shared_gpu_or_skip("chromakey_job_keys_green_with_ociobased_stub").is_none() + { return; } if oak_core::color::set_up_default_config().is_err() { @@ -650,8 +647,7 @@ fn chromakey_job_keys_green_with_ociobased_stub() { /// target-anchored behavior drew 50% vs 12.5%). #[test] fn shape_generator_size_is_sequence_relative() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("skipping shape_generator_size_is_sequence_relative: no GPU adapter"); + if oak_core::backend::shared_gpu_or_skip("shape_generator_size_is_sequence_relative").is_none() { return; } let path = clip_path("shape_seqrel"); @@ -693,9 +689,7 @@ fn shape_generator_size_is_sequence_relative() { PixelFormat::F32, ) .expect("shape render"); - let Texture::Cpu(frame) = &tex else { - panic!("graph render produced a non-CPU texture"); - }; + let frame = tex.to_frame().expect("shape frame readback"); let stride = frame.linesize_bytes() as usize; let y = (height / 2) as usize; let mut red = 0usize; diff --git a/crates/oak-render/tests/ofxmisc_blur.rs b/crates/oak-render/tests/ofxmisc_blur.rs index e2263148b..01618636a 100644 --- a/crates/oak-render/tests/ofxmisc_blur.rs +++ b/crates/oak-render/tests/ofxmisc_blur.rs @@ -4,7 +4,7 @@ use oak_core::{PixelFormat, Rational}; use oak_node::value::{NodeValue, NodeValueRow, NodeValueTable, ValueType}; fn texture_value(t: Texture) -> NodeValue { NodeValue::Texture(oak_node::handle::make_owned(t)) } -fn gpu() -> bool { oak_core::backend::GpuContext::shared().is_some() } +fn gpu() -> bool { oak_core::backend::shared_gpu_or_skip("an OFX-Misc GPU test").is_some() } fn filled_frame(size: (i32, i32), rgba: [f32; 4]) -> Texture { let mut f = oak_render::eval::generate_frame(Rational::new(0, 1), size, PixelFormat::F32).unwrap(); for px in f.data.chunks_exact_mut(16) { for (c, v) in px.chunks_exact_mut(4).zip(rgba) { c.copy_from_slice(&v.to_le_bytes()); } } diff --git a/crates/oak-render/tests/ofxmisc_color.rs b/crates/oak-render/tests/ofxmisc_color.rs index af8fafa57..9fb0d36ab 100644 --- a/crates/oak-render/tests/ofxmisc_color.rs +++ b/crates/oak-render/tests/ofxmisc_color.rs @@ -4,7 +4,7 @@ use oak_core::{PixelFormat, Rational}; use oak_node::value::{NodeValue, NodeValueRow, NodeValueTable, ValueType}; fn texture_value(t: Texture) -> NodeValue { NodeValue::Texture(oak_node::handle::make_owned(t)) } -fn gpu() -> bool { oak_core::backend::GpuContext::shared().is_some() } +fn gpu() -> bool { oak_core::backend::shared_gpu_or_skip("an OFX-Misc GPU test").is_some() } fn filled_frame(size: (i32, i32), rgba: [f32; 4]) -> Texture { let mut f = oak_render::eval::generate_frame(Rational::new(0, 1), size, PixelFormat::F32).unwrap(); for px in f.data.chunks_exact_mut(16) { for (c, v) in px.chunks_exact_mut(4).zip(rgba) { c.copy_from_slice(&v.to_le_bytes()); } } diff --git a/crates/oak-render/tests/ofxmisc_gen.rs b/crates/oak-render/tests/ofxmisc_gen.rs index 5a009d9f3..341ce237e 100644 --- a/crates/oak-render/tests/ofxmisc_gen.rs +++ b/crates/oak-render/tests/ofxmisc_gen.rs @@ -4,7 +4,7 @@ use oak_core::{PixelFormat, Rational}; use oak_node::value::{NodeValue, NodeValueRow, NodeValueTable, ValueType}; fn texture_value(t: Texture) -> NodeValue { NodeValue::Texture(oak_node::handle::make_owned(t)) } -fn gpu() -> bool { oak_core::backend::GpuContext::shared().is_some() } +fn gpu() -> bool { oak_core::backend::shared_gpu_or_skip("an OFX-Misc GPU test").is_some() } fn filled_frame(size: (i32, i32), rgba: [f32; 4]) -> Texture { let mut f = oak_render::eval::generate_frame(Rational::new(0, 1), size, PixelFormat::F32).unwrap(); for px in f.data.chunks_exact_mut(16) { for (c, v) in px.chunks_exact_mut(4).zip(rgba) { c.copy_from_slice(&v.to_le_bytes()); } } diff --git a/crates/oak-render/tests/ofxmisc_matrix.rs b/crates/oak-render/tests/ofxmisc_matrix.rs index 1a152d240..30264d849 100644 --- a/crates/oak-render/tests/ofxmisc_matrix.rs +++ b/crates/oak-render/tests/ofxmisc_matrix.rs @@ -4,7 +4,7 @@ use oak_core::{PixelFormat, Rational}; use oak_node::value::{NodeValue, NodeValueRow, NodeValueTable, ValueType}; fn texture_value(t: Texture) -> NodeValue { NodeValue::Texture(oak_node::handle::make_owned(t)) } -fn gpu() -> bool { oak_core::backend::GpuContext::shared().is_some() } +fn gpu() -> bool { oak_core::backend::shared_gpu_or_skip("an OFX-Misc GPU test").is_some() } fn filled_frame(size: (i32, i32), rgba: [f32; 4]) -> Texture { let mut f = oak_render::eval::generate_frame(Rational::new(0, 1), size, PixelFormat::F32).unwrap(); for px in f.data.chunks_exact_mut(16) { for (c, v) in px.chunks_exact_mut(4).zip(rgba) { c.copy_from_slice(&v.to_le_bytes()); } } diff --git a/crates/oak-render/tests/ofxmisc_merge.rs b/crates/oak-render/tests/ofxmisc_merge.rs index 03e3d9301..aafbdc070 100644 --- a/crates/oak-render/tests/ofxmisc_merge.rs +++ b/crates/oak-render/tests/ofxmisc_merge.rs @@ -4,7 +4,7 @@ use oak_core::{PixelFormat, Rational}; use oak_node::value::{NodeValue, NodeValueRow, NodeValueTable, ValueType}; fn texture_value(t: Texture) -> NodeValue { NodeValue::Texture(oak_node::handle::make_owned(t)) } -fn gpu() -> bool { oak_core::backend::GpuContext::shared().is_some() } +fn gpu() -> bool { oak_core::backend::shared_gpu_or_skip("an OFX-Misc GPU test").is_some() } fn filled_frame(size: (i32, i32), rgba: [f32; 4]) -> Texture { let mut f = oak_render::eval::generate_frame(Rational::new(0, 1), size, PixelFormat::F32).unwrap(); for px in f.data.chunks_exact_mut(16) { for (c, v) in px.chunks_exact_mut(4).zip(rgba) { c.copy_from_slice(&v.to_le_bytes()); } } diff --git a/crates/oak-render/tests/pipeline_test.rs b/crates/oak-render/tests/pipeline_test.rs index e90f97c15..97857f1b6 100644 --- a/crates/oak-render/tests/pipeline_test.rs +++ b/crates/oak-render/tests/pipeline_test.rs @@ -27,7 +27,7 @@ mod common; use oak_core::{PixelFormat, Rational}; -use oak_core::backend::{BackendKind, DisplayRenderer, GpuContext}; +use oak_core::backend::{BackendKind, DisplayRenderer}; use oak_core::frame::VideoParamsPod; use oak_core::texture::{Frame, Texture}; @@ -149,8 +149,7 @@ fn texture_roundtrip_bit_exact_f32() { /// backend; tolerance 1e-4 for driver variance. #[test] fn gpu_path_f32_invariants() { - let Some(ctx) = GpuContext::create(BackendKind::Auto) else { - eprintln!("no GPU adapter; skipping gpu_path_f32_invariants"); + let Some(ctx) = oak_core::backend::gpu_or_skip("gpu_path_f32_invariants") else { return; }; let w = 16; diff --git a/crates/oak-render/tests/render_threads_test.rs b/crates/oak-render/tests/render_threads_test.rs index 3185ba969..ec9651a9d 100644 --- a/crates/oak-render/tests/render_threads_test.rs +++ b/crates/oak-render/tests/render_threads_test.rs @@ -163,11 +163,11 @@ fn render_video(params: VideoTicketParams) -> Texture { } } -fn frame_of(texture: &Texture) -> &Frame { - let Texture::Cpu(frame) = texture else { - panic!("ticket produced a non-CPU texture"); - }; - frame +/// The frame bytes of a rendered texture. GPU textures (the pipeline +/// backend on a GPU-capable host) are read back for the assertion; the +/// playback path itself never downloads. +fn frame_of(texture: &Texture) -> Frame { + texture.to_frame().expect("ticket frame readback") } /// Byte-for-byte frame equality with a first-difference report. @@ -403,6 +403,174 @@ fn build_project(clips: &[(&str, Rational, Rational)]) -> (Arc>, (project, seq) } +/// Create one footage clip on `track` (not appended: the layered fixture +/// orders V1 as clip/transition/clip explicitly). +fn add_clip( + graph: &mut oak_node::graph::Graph, + path: &Path, + in_: Rational, + out: Rational, +) -> NodeId { + let mut footage = FootageBehavior::new(path.to_string_lossy().as_ref()); + footage.probe().expect("probe the generated clip"); + let footage = graph.add_node(NodeCore::new(), Box::new(footage)); + let (ccore, cbehavior) = clip_create(); + let clip = graph.add_node(ccore, cbehavior); + graph + .connect(footage, clip, clip_input::TEXTURE_INPUT, -1) + .expect("connect footage to clip"); + graph + .get_mut(clip) + .unwrap() + .behavior + .as_any_mut() + .unwrap() + .downcast_mut::() + .expect("clip block") + .core + .range = TimeRange::new(in_, out); + clip +} + +fn track_mut(graph: &mut oak_node::graph::Graph, id: NodeId) -> &mut TrackBehavior { + graph + .get_mut(id) + .unwrap() + .behavior + .as_any_mut() + .unwrap() + .downcast_mut::() + .expect("video track") +} + +/// The layered M2 playback fixture: V1 carries two clips joined by a +/// transition, V2 an overlapping clip (multi-track composite) and V3 an +/// adjustment layer with an Opacity effect (the sweep). At the transition +/// seam (t=1) one frame exercises all three mechanisms — the transitions' +/// two decoded sides, the multi-track composite and the adjustment sweep +/// — end to end. +fn build_layered_project( + first: &Path, + second: &Path, + below: &Path, +) -> (Arc>, NodeId) { + pin_legacy_working_space(); + let project = Project::new(); + let seq; + { + let mut p = project.lock().unwrap(); + p.initialize().expect("initialize the project"); + let (score, sbehavior) = SequenceBehavior::create(); + seq = p.graph.add_node(score, sbehavior); + let (tl_core, tl_beh) = TrackListBehavior::create(); + let tl = p.graph.add_node(tl_core, tl_beh); + + // V1: A [0,1) + transition [0.5,1.5) + B [1,2). + let (v1_core, v1_beh) = TrackBehavior::create(); + let v1 = p.graph.add_node(v1_core, v1_beh); + let a = add_clip(&mut p.graph, first, Rational::new(0, 1), Rational::new(1, 1)); + let b = add_clip(&mut p.graph, second, Rational::new(1, 1), Rational::new(2, 1)); + let (tcore, tbehavior) = oak_node::block::transition_create(); + let transition = p.graph.add_node(tcore, tbehavior); + { + let behavior = p + .graph + .get_mut(transition) + .unwrap() + .behavior + .as_any_mut() + .unwrap() + .downcast_mut::() + .expect("transition block"); + behavior.core.range = TimeRange::new(Rational::new(1, 2), Rational::new(3, 2)); + behavior.in_offset = Rational::new(1, 2); + behavior.out_offset = Rational::new(1, 2); + } + p.graph + .connect( + a, + transition, + oak_node::block::transition_input::OUT_BLOCK, + -1, + ) + .expect("connect the outgoing clip to the transition"); + p.graph + .connect( + b, + transition, + oak_node::block::transition_input::IN_BLOCK, + -1, + ) + .expect("connect the incoming clip to the transition"); + track_mut(&mut p.graph, v1).blocks = vec![a, transition, b]; + + // V2: C [0,2), overlapping the transition track. + let (v2_core, v2_beh) = TrackBehavior::create(); + let v2 = p.graph.add_node(v2_core, v2_beh); + let c = add_clip(&mut p.graph, below, Rational::new(0, 1), Rational::new(2, 1)); + track_mut(&mut p.graph, v2).append_block(c); + + // V3: an adjustment layer [0,2) with an Opacity(0.75) chain. + let (v3_core, v3_beh) = TrackBehavior::create(); + let v3 = p.graph.add_node(v3_core, v3_beh); + let (acore, abehavior) = oak_node::block::adjustment_create(); + let adjustment = p.graph.add_node(acore, abehavior); + p.graph + .get_mut(adjustment) + .unwrap() + .behavior + .as_any_mut() + .unwrap() + .downcast_mut::() + .expect("adjustment block") + .core + .range = TimeRange::new(Rational::new(0, 1), Rational::new(2, 1)); + let (ecore, ebehavior) = oak_node::nodes::opacity::create(); + let effect = p.graph.add_node(ecore, ebehavior); + p.graph + .connect( + effect, + adjustment, + oak_node::block::adjustment_input::TEXTURE_INPUT, + -1, + ) + .expect("connect opacity to the adjustment block"); + p.graph.get_mut(effect).unwrap().core.set_standard_value( + oak_node::nodes::opacity::VALUE_INPUT, + -1, + oak_node::value::NodeValue::Float(0.75), + ); + track_mut(&mut p.graph, v3).append_block(adjustment); + + // V3 is last = topmost: its sweep covers V1 and V2. + { + let tl = p + .graph + .get_mut(tl) + .unwrap() + .behavior + .as_any_mut() + .unwrap() + .downcast_mut::() + .expect("video track list"); + tl.tracks.push(v1); + tl.tracks.push(v2); + tl.tracks.push(v3); + } + p.graph + .get_mut(seq) + .unwrap() + .behavior + .as_any_mut() + .unwrap() + .downcast_mut::() + .expect("sequence") + .track_lists + .push(tl); + } + (project, seq) +} + /// `OAK_PIPELINE=threads` selects the thread pipeline (the default stays /// the process backend, which the manager-guard init below exercises as /// the test-only inline choice). @@ -549,6 +717,112 @@ fn pipeline_viewer_ticket_matches_inline_pixels() { let _ = std::fs::remove_file(&path); } +// M2: the playback graph path keeps every frame on the GPU. Rendering +// a sequence graph through the thread pipeline must transfer pixels +// CPU→GPU once per decoded frame (the M5 gap — decode is still CPU) +// and never read back; the final texture is GPU-resident until the +// presentation boundary. +#[test] +fn pipeline_graph_playback_has_zero_gpu_readbacks() { + let _lock = lock(); + pin_legacy_working_space(); + if oak_core::backend::shared_gpu_or_skip("the pipeline GPU zero-copy assertion").is_none() { + return; + } + let path = test_clip("gpu_zero"); + let filename = path.to_string_lossy().to_string(); + let clip = (filename.as_str(), Rational::new(0, 1), Rational::new(1, 1)); + let (project, sequence) = build_project(&[clip]); + let uuid = project.lock().unwrap().uuid.clone(); + let viewer = sequence.identity(); + let times = [ + Rational::new(0, 1), + Rational::new(3, 10), + Rational::new(6, 10), + ]; + + let guard = common::ManagerGuard::init_with(RenderBackendChoice::Pipeline); + let manager = RenderManager::global().expect("manager installed"); + manager.set_inline_project(project.clone()); + if let Some(backend) = manager.pipeline_backend() { + assert!(backend.decode_service().wait_idle(), "service drained"); + } + oak_core::backend::reset_gpu_transfer_counters(); + let mut rendered = 0u64; + for &time in × { + let texture = render_video(viewer_params(&uuid, viewer, time)); + assert!( + matches!(texture, Texture::Gpu { .. }), + "the thread-pipeline graph path must produce a GPU texture" + ); + rendered += 1; + } + let (uploads, downloads) = oak_core::backend::gpu_transfer_counters(); + assert_eq!(downloads, 0, "playback must not read the frame back to CPU"); + assert_eq!( + uploads, rendered, + "one decode upload per frame until M5 imports the decode surface" + ); + drop(guard); + let _ = std::fs::remove_file(&path); +} +/// M2: the layered playback path — multi-track composite + transition +/// blend + adjustment sweep — is zero-readback too. The single clip test +/// above covers the common case; this one proves the per-clip readback +/// pattern that used to exist in each of these paths is gone: every clip +/// uploads once (the M5 gap) and nothing comes back. +#[test] +fn pipeline_layered_playback_has_zero_gpu_readbacks() { + let _lock = lock(); + if oak_core::backend::shared_gpu_or_skip("the layered playback zero-readback assertion").is_none() + { + return; + } + let first = test_clip("layered_first"); + let second = test_clip_copy(&first, "layered_second"); + let below = test_clip_copy(&first, "layered_below"); + let (project, sequence) = build_layered_project(&first, &second, &below); + let uuid = project.lock().unwrap().uuid.clone(); + let viewer = sequence.identity(); + + let guard = common::ManagerGuard::init_with(RenderBackendChoice::Pipeline); + let manager = RenderManager::global().expect("manager installed"); + manager.set_inline_project(project.clone()); + if let Some(backend) = manager.pipeline_backend() { + assert!(backend.decode_service().wait_idle(), "service drained"); + } + oak_core::backend::reset_gpu_transfer_counters(); + // The transition seam: V1 blends A/B, V2 composites underneath and V3 + // sweeps the result with Opacity(0.75). + let texture = render_video(viewer_params(&uuid, viewer, Rational::new(1, 1))); + assert!( + matches!(texture, Texture::Gpu { .. }), + "layered playback must produce a GPU texture" + ); + let (uploads, downloads) = oak_core::backend::gpu_transfer_counters(); + assert_eq!( + downloads, 0, + "multi-track/transition/adjustment playback must not read back" + ); + assert_eq!( + uploads, 3, + "each decoded clip uploads exactly once (A, B, C); the passes are GPU→GPU" + ); + // The adjustment sweep must have participated: the final alpha is the + // Opacity(0.75) value (readback only for the assertion, after the + // counter sample above). + let frame = texture.to_frame().expect("frame readback"); + let alpha = f32::from_le_bytes(frame.data[12..16].try_into().unwrap()); + assert!( + alpha > 0.0 && alpha < 0.99, + "the adjustment sweep applied (alpha {alpha})" + ); + drop(guard); + let _ = std::fs::remove_file(&first); + let _ = std::fs::remove_file(&second); + let _ = std::fs::remove_file(&below); +} + /// A saturated render queue closes the decode service's prefetch gate: a /// speculative decode must be refused while a frame is in flight and the /// queue is full, and everything queued must still run once the in-flight diff --git a/crates/oak-render/tests/text_outline_glow.rs b/crates/oak-render/tests/text_outline_glow.rs index d24c0086e..0c311c843 100644 --- a/crates/oak-render/tests/text_outline_glow.rs +++ b/crates/oak-render/tests/text_outline_glow.rs @@ -10,7 +10,7 @@ use oak_node::nodes::textbackend::{ #[allow(dead_code)] fn texture_value(t: Texture) -> NodeValue { NodeValue::Texture(oak_node::handle::make_owned(t)) } -fn gpu() -> bool { oak_core::backend::GpuContext::shared().is_some() } +fn gpu() -> bool { oak_core::backend::shared_gpu_or_skip("a text outline/glow GPU test").is_some() } #[allow(dead_code)] fn filled_frame(size: (i32, i32), rgba: [f32; 4]) -> Texture { let mut f = oak_render::eval::generate_frame(Rational::new(0, 1), size, PixelFormat::F32).unwrap(); diff --git a/crates/oak-render/tests/transition_render.rs b/crates/oak-render/tests/transition_render.rs index d295d6315..b9c647425 100644 --- a/crates/oak-render/tests/transition_render.rs +++ b/crates/oak-render/tests/transition_render.rs @@ -218,11 +218,10 @@ fn set_style(project: &Arc>, block: NodeId, style: i64) { } /// The raw CPU frame bytes of a rendered texture. -fn frame_data(texture: &Texture) -> &[u8] { - let Texture::Cpu(frame) = texture else { - panic!("graph render produced a non-CPU texture"); - }; - &frame.data +/// The raw frame bytes of a rendered texture (GPU textures are read back +/// for the assertion; the playback path itself never downloads). +fn frame_data(texture: &Texture) -> Vec { + texture.to_frame().expect("graph frame readback").data } /// Render one 64x64 F32 frame of `seq` at `time` and return its bytes. @@ -230,7 +229,7 @@ fn render_frame(project: &Arc>, seq: NodeId, time: Rational) -> V let texture = oak_render::eval::render_graph_frame(project, seq, time, (64, 64), PixelFormat::F32) .expect("graph render"); - frame_data(&texture).to_vec() + frame_data(&texture) } /// The F32 RGBA channel of a 64x64 frame at `(x, y)`. @@ -261,8 +260,9 @@ fn channel_mean(data: &[u8], c: usize, xs: &[usize], ys: &[usize]) -> f32 { /// Skipped (with a note) when no GPU adapter exists. #[test] fn cross_dissolve_blends_the_two_sides_of_a_cut() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("skipping cross_dissolve_blends_the_two_sides_of_a_cut: no GPU adapter"); + if oak_core::backend::shared_gpu_or_skip("cross_dissolve_blends_the_two_sides_of_a_cut") + .is_none() + { return; } let red = clip_path("dissolve_red"); @@ -344,8 +344,9 @@ fn cross_dissolve_blends_the_two_sides_of_a_cut() { /// outgoing one on the right (the outgoing image leads the sweep). #[test] fn wipe_style_splits_the_frame_at_the_boundary() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("skipping wipe_style_splits_the_frame_at_the_boundary: no GPU adapter"); + if oak_core::backend::shared_gpu_or_skip("wipe_style_splits_the_frame_at_the_boundary") + .is_none() + { return; } let red = clip_path("wipe_red"); @@ -402,8 +403,9 @@ fn wipe_style_splits_the_frame_at_the_boundary() { /// same shader with the open side generated transparent. #[test] fn single_sided_transitions_fade_from_and_to_black() { - if oak_core::backend::GpuContext::shared().is_none() { - eprintln!("skipping single_sided_transitions_fade_from_and_to_black: no GPU adapter"); + if oak_core::backend::shared_gpu_or_skip("single_sided_transitions_fade_from_and_to_black") + .is_none() + { return; } let red = clip_path("edge_red"); diff --git a/crates/oak-render/tests/transitionfx.rs b/crates/oak-render/tests/transitionfx.rs index 608ef2c20..f2eb720ee 100644 --- a/crates/oak-render/tests/transitionfx.rs +++ b/crates/oak-render/tests/transitionfx.rs @@ -27,7 +27,7 @@ use oak_node::value::{NodeValue, NodeValueRow, NodeValueTable, ValueType}; const TRANSITIONFX: &str = "org.olivevideoeditor.Olive.transitionfx"; fn texture_value(t: Texture) -> NodeValue { NodeValue::Texture(oak_node::handle::make_owned(t)) } -fn gpu() -> bool { oak_core::backend::GpuContext::shared().is_some() } +fn gpu() -> bool { oak_core::backend::shared_gpu_or_skip("a transition effect test").is_some() } fn filled_frame(size: (i32, i32), rgba: [f32; 4]) -> Texture { let mut f = oak_render::eval::generate_frame(Rational::new(0, 1), size, PixelFormat::F32).unwrap(); for px in f.data.chunks_exact_mut(16) { for (c, v) in px.chunks_exact_mut(4).zip(rgba) { c.copy_from_slice(&v.to_le_bytes()); } } diff --git a/crates/oak-task/src/export.rs b/crates/oak-task/src/export.rs index eedf11237..973b309f4 100644 --- a/crates/oak-task/src/export.rs +++ b/crates/oak-task/src/export.rs @@ -279,16 +279,19 @@ impl ExportTask { } } - /// Copy a rendered `oakrender` CPU texture into an `oakcodec` frame - /// with the matching video params (row-wise copy — line sizes may - /// differ between the render and codec frame layouts). + /// Copy a rendered `oakrender` texture into an `oakcodec` frame with + /// the matching video params (row-wise copy — line sizes may differ + /// between the render and codec frame layouts). A GPU texture is read + /// back here: the encoder input is one of the three explicit CPU + /// boundaries (M2). fn to_codec_frame(texture: &Texture) -> Result { - let Texture::Cpu(frame) = texture else { - return Err(Error::Failed( - "Render produced a GPU texture; the CPU encoder path cannot consume it" - .to_string(), - )); + let frame = match texture { + Texture::Cpu(frame) => frame.clone(), + Texture::Gpu { .. } => texture.to_frame().map_err(|e| { + Error::Failed(format!("Render frame readback for the encoder failed: {e:?}")) + })?, }; + let frame = &frame; let params = CommonVideoParams::new_basic( frame.width, frame.height, diff --git a/crates/oak-worker/src/worker.rs b/crates/oak-worker/src/worker.rs index 8fd7bf61e..c87e3a7e1 100644 --- a/crates/oak-worker/src/worker.rs +++ b/crates/oak-worker/src/worker.rs @@ -1280,24 +1280,39 @@ fn render_f32_into( match viewer_id { Some(viewer_id) => { let rendered = eval::render_graph_frame(project, viewer_id, time, (w, h), PixelFormat::F32); - match &rendered { - Ok(oak_core::texture::Texture::Cpu(frame)) => { - let src_stride = frame.linesize_bytes() as usize; - let row_bytes = (w as usize) * 16; - if frame.data.len() < src_stride * (h as usize) - || dst.len() < row_bytes * (h as usize) - { - return Err("graph frame geometry mismatch".to_string()); + // M2: the graph renders all-GPU in-process; the worker's + // wire format is a CPU shm slot, so this is the explicit + // readback boundary of the process backend. + let frame = match rendered { + Ok(oak_core::texture::Texture::Cpu(frame)) => Some(frame), + Ok(texture @ oak_core::texture::Texture::Gpu { .. }) => { + match texture.to_frame() { + Ok(frame) => Some(frame), + Err(e) => { + warn_graph_fallback(spec.viewer_node, &e.to_string()); + None + } } - for y in 0..h as usize { - dst[y * row_bytes..(y + 1) * row_bytes].copy_from_slice( - &frame.data[y * src_stride..y * src_stride + row_bytes], - ); - } - return Ok(()); } - Ok(_) => return Err("graph render produced a GPU texture".to_string()), - Err(e) => warn_graph_fallback(spec.viewer_node, &e.to_string()), + Err(e) => { + warn_graph_fallback(spec.viewer_node, &e.to_string()); + None + } + }; + if let Some(frame) = frame { + let src_stride = frame.linesize_bytes() as usize; + let row_bytes = (w as usize) * 16; + if frame.data.len() < src_stride * (h as usize) + || dst.len() < row_bytes * (h as usize) + { + return Err("graph frame geometry mismatch".to_string()); + } + for y in 0..h as usize { + dst[y * row_bytes..(y + 1) * row_bytes].copy_from_slice( + &frame.data[y * src_stride..y * src_stride + row_bytes], + ); + } + return Ok(()); } } None => warn_graph_fallback(spec.viewer_node, "viewer node not in graph"), @@ -1320,8 +1335,13 @@ fn render_f32_into( PixelFormat::F32, ) .map_err(|e| format!("footage decode: {e}"))?; - let oak_core::texture::Texture::Cpu(frame) = &decoded else { - return Err("decode produced a GPU texture".to_string()); + // M2: decode stays CPU for now (M5 makes it GPU); an imported GPU + // texture would still have to cross into the shm slot here. + let frame = match &decoded { + oak_core::texture::Texture::Cpu(frame) => frame.clone(), + gpu @ oak_core::texture::Texture::Gpu { .. } => gpu + .to_frame() + .map_err(|e| format!("decode readback: {e}"))?, }; let src_stride = frame.linesize_bytes() as usize; let row_bytes = (w as usize) * 16; diff --git a/docs/zh/plans/render-pipeline-threads.md b/docs/zh/plans/render-pipeline-threads.md index 7561eba62..ab8406af4 100644 --- a/docs/zh/plans/render-pipeline-threads.md +++ b/docs/zh/plans/render-pipeline-threads.md @@ -239,6 +239,51 @@ - CPU 边界的显式回读点只有三处:CPU OFX 插件(§3.2)、导出编码器输入、 磁盘帧缓存写入(FrameHashCache::SaveCacheFrame 对应物)。 +> **M2 攻关结论(2026-09-11 回填)**:共享 device 路线可行且已落地,前提是 +> **引擎与 gpui 统一到同一个 wgpu 大版本**。攻关发现: +> +> 1. gpui 的 `Window::gpu_context()`(Linux/FreeBSD)确实暴露窗口的 +> `(Arc, Arc)`,但 vendored gpui_wgpu 用的是 +> **wgpu 29**,而引擎此前是 **wgpu 25**——两个大版本的 `wgpu::Texture` +> 是不同类型,纹理无法跨越。M2 把 `oak-core`/`oak-render` 升到 +> **wgpu 29 + naga 29**(`backend.rs` 的 8 处破坏性 API 改动;其余代码 +> 只经 `GpuContext`),版本鸿沟消除。 +> 2. `GpuContext::adopt(device, queue, kind)` 采用宿主 device; +> `register_context`(`oak-app/src/oakui/gpu.rs`)在窗口建立时调用 +> `install_shared` 把它装进进程级 shared 槽,渲染线程因此在预览器同一 +> device 上出帧。`GpuContext::texture_handle` 把引擎纹理的 +> `Arc` 交给 gpui 的 `SurfaceSource::Texture`,上屏零拷贝。 +> 3. **色彩管理必须留在链上**:GPU 路径不能跳过 output node 与显示器 ICC。 +> 做法是把「工作空间 → 输出规格(`colormath::working_to_display_target`) +> → 显示器 ICC(`displaycolor::apply_f32_rgba`)」在 CPU 上用**原有精确 +> 实现**烘焙成 65³ 3D LUT(`oak-core::lut::Lut3d`,域 +> `[-0.25, 4]³`),经 `GpuContext::set_display_lut` 上传为 GPU 3D 纹理, +> 由 `present_texture` 的 WGSL pass 做手工三线性插值(不依赖 +> `FLOAT32_FILTERABLE`)。设置/显示器/ICC 变化(`displaycolor::generation` +> 或项目色彩设置)时重建。CPU 路径的 `apply_f32_rgba` 一行未动,GPU 与 +> CPU 逐点一致(测试对拍,f16 输出量化内)。同一套 LUT 机制也用于图内 +> `ColorTransformJob`:`process_color_transform_job` 对 GPU 纹理把 OCIO +> processor 经 CPU 参考烘焙成 3D LUT,用 `GpuContext::apply_color_lut` +> 在 GPU 上应用(`color_transform_lut` 按 processor cache id 缓存),不再 +> pass-through;只有无法烘焙时才走一次显式回读。 +> 4. **平台边界**:macOS/Windows 的 gpui 暂不暴露 device(macOS 走 +> `oak_bridge` IOSurface 的未来路线,见 acescg 计划 P2),此时 +> `present_gpu_frame` 返回 `None`,`to_display` 走**单点显式回读** +> (唯一一次 download,之后仍 CPU 上传到 gpui)。Linux 上若引擎 +> device 不是 adopted(例如进程池 worker 的独立 device),同样回退这条 +> 单点路径。CPU OFX、导出编码器、磁盘缓存三处边界显式回读不变。 +> `install_shared` 对「引擎已创建但尚未创建任何 GPU 资源」的上下文 +> 允许被 UI 设备**替换**(时序守卫:开窗前任何 `shared()` 触碰都不会 +> 静默丢掉零拷贝上屏),用过的上下文拒绝替换并记一条错误日志。 +> 5. 进程池后端(默认)仍在 worker 内完成图求值后**显式回读**成 shm 槽 +> (oak-worker/worker.rs 的 graph 分支),这是进程模型的必然;M4 通过 +> 验收前进程池仍是默认,线程管线(`OAK_PIPELINE=threads`)才走上述 +> 零拷贝上屏。 +> 6. 已知取舍:GPU 帧没有 `Frame::timestamp`(GPU 路径不需要);scopes/ +> 取色器的 CPU 兜底图是 1×1 占位(需要时可作为显式回读点按需填充); +> 上屏每帧新建一张 `Rgba16Float` 目标纹理(与 gpui 的取帧生命周期一致, +> 后续可做纹理环)。 + ### 3.6 平台互操作分支(解码零拷贝与上屏,用户硬性要求) **解码必须尽可能 GPU,并与渲染共用同一片 GPU 内存;CPU 解码后上传只作为 @@ -340,7 +385,7 @@ fallback。** 解码上传与上屏共用一层 `gpuinteop` 抽象,按后端 | **M0a Job 枚举化 + 单循环 resolve** | §3.7 全量:Job 枚举补全(含 CacheJob)并挂进输出表、resolve 单循环 match、子 job 递归 resolve | 全 workspace 测试绿;新增 CacheJob 磁盘缓存往返测试;`resolve_*_jobs` 四函数删除 | | **M0b Job 图 + 虚拟端点 + BFS** | §3.8 全量:图固定 GraphInput/GraphOutput 虚拟节点(默认相连、禁删禁复制、序列化往返)、节点编辑器显示两节点、resolve 改为从输入节点的 Kahn 形态 BFS | 新增测试:多输入汇合等齐全部输入、多输出分叉各自成帧、非全连通图不可达节点不执行、环报错断支、虚拟节点删除/复制被拒、序列化往返后端点仍在;节点编辑器 UI 测试(端点可见、入线/出线规则);既有测试全绿 | | **M1 线程管线骨架** | 解码/渲染/上屏三线程+三队列进 oak-render(`pipeline` 模块);RenderManager 增加线程后端,进程池后端保留,`OAK_PIPELINE=processes` 可回退 | 同一套渲染测试在两个后端下都绿(测试矩阵化);播放/seek/导出 smoke 等价 | -| **M2 GPU 零拷贝** | 图内全程 `Texture::Gpu`;上屏互操作攻关(§3.5)落地;导出/缓存/OFX 三处边界显式回读;内置 YUV→RGB GPU pass 替代 CPU swscale | 播放路径 GPU↔CPU 搬运次数为 0(计数断言,参照 M15 S2 的 `main_heap_frame_copies` 范式);`to_display` 不再接收 CPU 帧 | +| **M2 GPU 零拷贝** | 图内全程 `Texture::Gpu`(合成/转场/调整层不再逐帧回读);wgpu 29 统一 + 采用 gpui device(§3.5 攻关已回填);GPU 色彩管理(工作空间→输出规格→显示器 ICC 烘焙 3D LUT,GPU 执行);导出/缓存/OFX 三处边界显式回读;内置 YUV→RGB GPU pass(M5 解码导入的依赖项,解码接线随 M5) | 图播放路径 **GPU→CPU 回读为 0**(`oak_core::backend::gpu_transfer_counters` 计数断言,M1 帧缓存范式);`RenderedFrame::Gpu` + `to_display` 上屏在 adopted device 上零拷贝(app 测试);YUV→RGB pass 与 `colormath::yuv444p16_to_rgb_f32` 对拍;全 workspace 测试绿 | | **M3 OFX 独立进程** | oak-ofx-host 单进程宿主;PluginJob 经 IPC;崩溃重生+紫帧回退;进度/取消协议搬运 | 杀掉 ofx-host 进程 → 在途 job 重投成功;连续三次崩溃 → 紫帧;进度条/取消行为与现状一致 | | **M4 流水线预取** | 调度层按 §3.4 投依赖窗口;背压策略 | 1080p 播放 CPU 占用不升、fps 不低于进程池后端;首帧延迟不劣化(基准对比留档) | | **M5 GPU 解码零拷贝** | §3.6 表逐行落地:staging fallback 基线 → Linux NVDEC/VAAPI 导入 → Windows D3D11VA 导入 → macOS VideoToolbox 导入;FFmpeg 无 hwaccel 的组合才评估手写 GPU 解码 | 硬解路径 `HW_TRANSFERS` 计数归零(不再下载);逐平台导入开/关对比测试;每行独立 PR 可回退 | @@ -371,9 +416,12 @@ M1(与 M2 可并行);M4 依赖 M2;M5 依赖 M2(YUV→RGB pass 与互 ## 6. 风险与对策 -1. **上屏互操作不确定**(gpui 的 wgpu device 能否共享):M2 第一个工作项 - 就是攻关并回填结论;最坏情况退回"渲染线程 blit 到共享纹理"或"单点 - staging",损失一次拷贝而非架构。 +1. **上屏互操作不确定**(gpui 的 wgpu device 能否共享):**M2 已攻关并落地** + (结论见 §3.5 回填):把引擎从 wgpu 25 升到 29 后,`register_context` + 采用 gpui 的 device,`Texture::Gpu` 的原始纹理经 + `SurfaceSource::Texture` 直通 gpui,上屏零拷贝;颜色由 CPU 烘焙的 3D LUT + 在 GPU 应用,不跳过色彩管理。macOS/Windows 的 gpui 暂不暴露 device, + 自动回退到单点 staging(仍只此一处)。 2. **解码器线程安全性**:oak-codec 会话当前按进程级互斥共享,集中到一个 线程后语义更简单,但 hwaccel 解码上下文可能有线程亲和(VAAPI/NVDEC), M1 先做软解路径,hwaccel 随 M5 逐项验证。