From aae8159bd771d52ad6822db6207655bee606820e Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Tue, 29 Sep 2026 11:02:04 -0400 Subject: [PATCH 1/8] [INITIAL] Recreate the Vulkan transformer and conformance stack with ghstack [ghstack-poisoned] --- .ci/scripts/setup-vulkan-linux-deps.sh | 47 +++++++++++++------------- .ci/scripts/test_backend.sh | 11 ++---- .github/workflows/vulkan.yml | 30 ++++++---------- 3 files changed, 36 insertions(+), 52 deletions(-) diff --git a/.ci/scripts/setup-vulkan-linux-deps.sh b/.ci/scripts/setup-vulkan-linux-deps.sh index debe610a18a..c1981d5050b 100755 --- a/.ci/scripts/setup-vulkan-linux-deps.sh +++ b/.ci/scripts/setup-vulkan-linux-deps.sh @@ -67,35 +67,37 @@ install_vulkan_loader() { # libvulkan.so.1 (the Khronos loader that volk dlopen()s at runtime) is not part # of the NVIDIA driver and is absent from the CUDA builder image; vulkan-tools # provides vulkaninfo for the device sanity check. Both ship as native el8 RPMs. + # The NVIDIA ICD also needs EGL and X11 client libraries on headless runners. if command -v dnf >/dev/null 2>&1; then - _maybe_sudo dnf install -y vulkan-loader vulkan-tools + _maybe_sudo dnf install -y vulkan-loader vulkan-tools libglvnd-egl libX11 libXext fi } _find_nvidia_vulkan_library() { - # NVIDIA implements its Vulkan ICD inside libGLX_nvidia.so.0. The NVIDIA - # container runtime mounts this library into the container (it is pulled from - # the driver's ldcache when NVIDIA_DRIVER_CAPABILITIES includes graphics/all), - # so prefer ldconfig and fall back to the usual mount locations. - local lib cand - lib="$(ldconfig -p 2>/dev/null | awk '/libGLX_nvidia\.so\.0/ {print $NF; exit}')" - if [ -z "${lib}" ]; then - for cand in /usr/lib64/libGLX_nvidia.so.0 \ - /usr/lib/x86_64-linux-gnu/libGLX_nvidia.so.0 \ - /usr/lib/libGLX_nvidia.so.0; do + # NVIDIA provides EGL and GLX Vulkan ICDs. Prefer EGL on headless runners; + # the GLX entry point can fail to initialize without an X server. + local soname lib cand + for soname in libEGL_nvidia.so.0 libGLX_nvidia.so.0; do + lib="$(ldconfig -p 2>/dev/null | awk -v name="${soname}" '$1 == name {print $NF; exit}')" + if [ -n "${lib}" ]; then + printf '%s' "${lib}" + return + fi + for cand in "/usr/lib64/${soname}" \ + "/usr/lib/x86_64-linux-gnu/${soname}" \ + "/usr/lib/${soname}"; do if [ -e "${cand}" ]; then - lib="${cand}" - break + printf '%s' "${cand}" + return fi done - fi - printf '%s' "${lib}" + done } _vulkan_has_real_device() { # True if the loader enumerates a hardware GPU. vulkaninfo can exit non-zero # for unrelated reasons (no display/WSI), so key off the reported deviceType. - command -v vulkaninfo >/dev/null 2>&1 || return 0 + command -v vulkaninfo >/dev/null 2>&1 || return 1 vulkaninfo --summary 2>/dev/null | grep -qE 'PHYSICAL_DEVICE_TYPE_(DISCRETE|INTEGRATED|VIRTUAL)_GPU' } @@ -131,25 +133,24 @@ JSON echo "Real NVIDIA GPU selected; pinned Vulkan ICD to ${nvidia_lib}" return fi - echo "WARNING: ${nvidia_lib} present but no GPU enumerated; using SwiftShader." - # Surface why the NVIDIA driver did not enumerate (e.g. a missing dependency - # of libGLX_nvidia, or no render node) so the fallback is diagnosable in CI. + echo "ERROR: ${nvidia_lib} present but no GPU enumerated." + # Surface missing ICD dependencies and device-enumeration errors. + ldd "${nvidia_lib}" || true if command -v vulkaninfo >/dev/null 2>&1; then echo "--- NVIDIA Vulkan ICD diagnostic ---" VK_LOADER_DEBUG=warn vulkaninfo --summary 2>&1 | head -40 || true echo "--- end diagnostic ---" fi - unset VK_ICD_FILENAMES else - echo "WARNING: no NVIDIA Vulkan driver library found; using SwiftShader." + echo "ERROR: no NVIDIA Vulkan driver library found." fi - install_swiftshader + return 1 } VULKAN_SDK_VERSION="1.4.321.1" # The no-argument default installs SwiftShader so the existing CPU-runner CI is -# unchanged. Pass "real-gpu" to prefer a real system ICD when one is present. +# unchanged. Pass "real-gpu" to require a real system ICD. case "${1:-swiftshader}" in real-gpu) # Do not download the LunarG SDK here: its prebuilt glslc cannot run on the diff --git a/.ci/scripts/test_backend.sh b/.ci/scripts/test_backend.sh index 068d5adb260..8bfe33333a8 100755 --- a/.ci/scripts/test_backend.sh +++ b/.ci/scripts/test_backend.sh @@ -54,15 +54,8 @@ if [[ "$FLOW" == *qnn* ]]; then fi if [[ "$FLOW" == *vulkan* ]]; then - # Setup the Vulkan SDK and select an ICD: use the real system GPU ICD when one - # is present (real-GPU runner), otherwise fall back to SwiftShader (CPU - # runner). The Vulkan loader searches both standard ICD directories. - if ls /etc/vulkan/icd.d/*.json /usr/share/vulkan/icd.d/*.json \ - >/dev/null 2>&1; then - source .ci/scripts/setup-vulkan-linux-deps.sh "real-gpu" - else - source .ci/scripts/setup-vulkan-linux-deps.sh "swiftshader" - fi + # CPU runners can have Mesa ICDs installed without a usable hardware GPU. + source .ci/scripts/setup-vulkan-linux-deps.sh "swiftshader" EXTRA_BUILD_ARGS+=" -DEXECUTORCH_BUILD_VULKAN=ON" fi diff --git a/.github/workflows/vulkan.yml b/.github/workflows/vulkan.yml index 2514a8d582c..dd3de1f33fb 100644 --- a/.github/workflows/vulkan.yml +++ b/.github/workflows/vulkan.yml @@ -70,10 +70,7 @@ jobs: # not run, so setup-vulkan-linux-deps.sh sources those from conda-forge and # the system package manager instead. The NVIDIA container runtime mounts # the driver's Vulkan library but not its ICD manifest, so the script - # synthesizes one and pins the loader to it; if no NVIDIA library is found - # it falls back to SwiftShader. - # NOTE: first-run check - inspect the vulkaninfo output below to confirm a - # real NVIDIA device is selected (not llvmpipe/SwiftShader). + # synthesizes one and requires the loader to enumerate a hardware GPU. source .ci/scripts/setup-vulkan-linux-deps.sh real-gpu vulkaninfo --summary || true @@ -100,23 +97,16 @@ jobs: # Operator coverage (mirrors test-vulkan-operators-linux, on real hardware). # The custom-op prototyping binaries are GPU microbenchmarks that rely on - # GPU timestamp queries; they need a real device and crash on the - # SwiftShader software fallback. Always build them (compile coverage), but - # only run them when a real GPU was selected (setup-vulkan-linux-deps.sh - # exports ETVK_USING_SWIFTSHADER when it falls back to SwiftShader). + # GPU timestamp queries, so they require the hardware selected above. PYTHON_EXECUTABLE=python bash backends/vulkan/test/custom_ops/build_and_run.sh - if [ -z "${ETVK_USING_SWIFTSHADER:-}" ]; then - ./cmake-out/backends/vulkan/test/custom_ops/test_add - ./cmake-out/backends/vulkan/test/custom_ops/test_q8csw_linear - ./cmake-out/backends/vulkan/test/custom_ops/test_q8csw_conv2d - ./cmake-out/backends/vulkan/test/custom_ops/test_q4gsw_linear - ./cmake-out/backends/vulkan/test/custom_ops/test_choose_qparams_per_row - ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_qdq - ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_clone - ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_binary - else - echo "SwiftShader fallback active: built custom-op benchmarks but skipping execution (they require real-GPU timestamp queries)." - fi + ./cmake-out/backends/vulkan/test/custom_ops/test_add + ./cmake-out/backends/vulkan/test/custom_ops/test_q8csw_linear + ./cmake-out/backends/vulkan/test/custom_ops/test_q8csw_conv2d + ./cmake-out/backends/vulkan/test/custom_ops/test_q4gsw_linear + ./cmake-out/backends/vulkan/test/custom_ops/test_choose_qparams_per_row + ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_qdq + ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_clone + ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_binary PYTHON_EXECUTABLE=python bash backends/vulkan/test/scripts/test_op.sh --build From bcab9d05737d396af7166c882baf26985f78cc56 Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Tue, 29 Sep 2026 11:02:04 -0400 Subject: [PATCH 2/8] [INITIAL] Recreate the Vulkan transformer and conformance stack with ghstack [ghstack-poisoned] --- .github/workflows/pull.yml | 1 + .../serialization/vulkan_graph_builder.py | 8 +----- backends/vulkan/test/targets.bzl | 11 ++++++++ .../vulkan/test/test_vulkan_graph_builder.py | 28 +++++++++++++++++++ 4 files changed, 41 insertions(+), 7 deletions(-) diff --git a/.github/workflows/pull.yml b/.github/workflows/pull.yml index 2f91799b9a4..a70a824f223 100644 --- a/.github/workflows/pull.yml +++ b/.github/workflows/pull.yml @@ -1682,6 +1682,7 @@ jobs: # route in the future. python -m unittest backends/vulkan/test/test_vulkan_delegate.py -k "*pt2e*" python -m unittest backends/vulkan/test/test_vulkan_delegate.py -k "*torchao*" + python -m unittest backends/vulkan/test/test_vulkan_graph_builder.py test-coreml-bc-macos: needs: [changed-files, run-decision] diff --git a/backends/vulkan/serialization/vulkan_graph_builder.py b/backends/vulkan/serialization/vulkan_graph_builder.py index 3de60966422..8703d391ce0 100644 --- a/backends/vulkan/serialization/vulkan_graph_builder.py +++ b/backends/vulkan/serialization/vulkan_graph_builder.py @@ -231,13 +231,7 @@ def create_null_value(self) -> int: return new_id def get_or_create_scalar_value(self, scalar: _ScalarType) -> int: - scalar_key = scalar - # Since Python considers 1 and True to be "equivalent" (as well as 0 and False) - # to distinguish entries in the dictionary, if scalar is bool then convert it - # to a string representation to use as a key for the dictionary - if isinstance(scalar, bool): - scalar_key = str(scalar) - + scalar_key = (type(scalar), repr(scalar)) if scalar_key in self.const_scalar_to_value_ids: return self.const_scalar_to_value_ids[scalar_key] diff --git a/backends/vulkan/test/targets.bzl b/backends/vulkan/test/targets.bzl index 77e74f9c8fc..5734e733195 100644 --- a/backends/vulkan/test/targets.bzl +++ b/backends/vulkan/test/targets.bzl @@ -29,6 +29,17 @@ def define_common_targets(is_fbcode = False): ], ) + python_unittest( + name = "test_vulkan_graph_builder", + srcs = ["test_vulkan_graph_builder.py"], + deps = [ + "//caffe2:torch", + "//executorch/backends/vulkan/serialization:lib", + "//executorch/backends/vulkan:vulkan_preprocess", + "//executorch/exir:lib", + ], + ) + python_unittest( name = "test_vulkan_passes", srcs = [ diff --git a/backends/vulkan/test/test_vulkan_graph_builder.py b/backends/vulkan/test/test_vulkan_graph_builder.py index 65afc3a2542..c180308e77a 100644 --- a/backends/vulkan/test/test_vulkan_graph_builder.py +++ b/backends/vulkan/test/test_vulkan_graph_builder.py @@ -8,12 +8,40 @@ import torch from executorch.backends.vulkan.serialization.vulkan_graph_builder import VkGraphBuilder +from executorch.backends.vulkan.serialization.vulkan_graph_schema import ( + Bool, + Double, + Int, +) from executorch.backends.vulkan.vulkan_preprocess import apply_passes from executorch.exir import to_edge from executorch.exir.backend.utils import DelegateMappingBuilder from executorch.exir.passes import SpecPropPass +class TestVkGraphBuilderScalarTensor(unittest.TestCase): + def test_scalar_cache_preserves_types_and_signed_zero(self): + program = torch.export.export(torch.nn.Identity(), (torch.ones(1),)) + builder = VkGraphBuilder( + program, DelegateMappingBuilder(generated_identifiers=True) + ) + scalars = (1.0, 1, True, 0.0, -0.0, 0, False) + expected = ( + Double(1.0), + Int(1), + Bool(True), + Double(0.0), + Double(-0.0), + Int(0), + Bool(False), + ) + ids = [builder.get_or_create_scalar_value(value) for value in scalars] + self.assertEqual(len(set(ids)), len(scalars)) + for value, value_id, serialized in zip(scalars, ids, expected): + self.assertEqual(builder.get_or_create_scalar_value(value), value_id) + self.assertEqual(repr(builder.values[value_id].value), repr(serialized)) + + class TestVkGraphBuilderInputIds(unittest.TestCase): """The serialized input list has to match the delegate call's arguments. From d6db604f7278972b5d8ce1a27323e738a25d4690 Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Tue, 29 Sep 2026 11:02:04 -0400 Subject: [PATCH 3/8] [INITIAL] Recreate the Vulkan transformer and conformance stack with ghstack [ghstack-poisoned] --- backends/vulkan/serialization/vulkan_graph_serialize.py | 4 ++-- backends/vulkan/test/test_serialization.py | 8 ++++++++ 2 files changed, 10 insertions(+), 2 deletions(-) diff --git a/backends/vulkan/serialization/vulkan_graph_serialize.py b/backends/vulkan/serialization/vulkan_graph_serialize.py index 665c23b8df2..b20073493af 100644 --- a/backends/vulkan/serialization/vulkan_graph_serialize.py +++ b/backends/vulkan/serialization/vulkan_graph_serialize.py @@ -25,7 +25,7 @@ VkGraph, ) from executorch.exir._serialize._dataclass import _DataclassEncoder, _json_to_dataclass -from executorch.exir._serialize._flatbuffer import _flatc_compile, _flatc_decompile +from executorch.exir._serialize._flatbuffer import _flatc_decompile, _run_flatc # Python's json module spells the non-finite floats "Infinity" / "-Infinity" / @@ -110,7 +110,7 @@ def convert_to_flatbuffer(vk_graph: VkGraph) -> bytes: json_path = os.path.join(d, "schema.json") with open(json_path, "wb") as json_file: json_file.write(vk_graph_json.encode("ascii")) - _flatc_compile(d, schema_path, json_path) + _run_flatc(["--binary", "--force-defaults", "-o", d, schema_path, json_path]) output_path = os.path.join(d, "schema.bin") with open(output_path, "rb") as output_file: return output_file.read() diff --git a/backends/vulkan/test/test_serialization.py b/backends/vulkan/test/test_serialization.py index 540b86ace82..8b95b1df2a1 100644 --- a/backends/vulkan/test/test_serialization.py +++ b/backends/vulkan/test/test_serialization.py @@ -372,6 +372,14 @@ def test_serialize_deserialize_non_finite_scalars(self) -> None: ] ) + def test_serialize_deserialize_signed_zero(self) -> None: + values = [ + VkValue(Double(-0.0)), + VkValue(Double(0.0)), + VkValue(DoubleList([-0.0, 0.0])), + ] + self.assertEqual(repr(self._round_trip(values).values), repr(values)) + def test_serialize_deserialize_non_finite_floats_in_list(self) -> None: # json only emits a float as a chunk of its own inside an object; in a # list the chunk carries the delimiter with it, so a rewrite that works From 446a5d62fbe307f11d2d80474ce4870c10938c7a Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Tue, 29 Sep 2026 11:02:05 -0400 Subject: [PATCH 4/8] [INITIAL] Recreate the Vulkan transformer and conformance stack with ghstack [ghstack-poisoned] --- .github/workflows/pull.yml | 1 + .github/workflows/vulkan.yml | 2 + .../_passes/squeeze_unsqueeze_inputs.py | 4 +- .../runtime/graph/ops/glsl/activations.h | 19 ++ .../runtime/graph/ops/glsl/unary_op.yaml | 2 + .../vulkan/runtime/graph/ops/impl/UnaryOp.cpp | 12 +- backends/vulkan/test/op_tests/cases.py | 9 +- backends/vulkan/test/targets.bzl | 18 ++ backends/vulkan/test/test_vulkan_dynamic.py | 166 ++++++++++++++++++ 9 files changed, 223 insertions(+), 10 deletions(-) create mode 100644 backends/vulkan/test/test_vulkan_dynamic.py diff --git a/.github/workflows/pull.yml b/.github/workflows/pull.yml index a70a824f223..8c7c5108395 100644 --- a/.github/workflows/pull.yml +++ b/.github/workflows/pull.yml @@ -1683,6 +1683,7 @@ jobs: python -m unittest backends/vulkan/test/test_vulkan_delegate.py -k "*pt2e*" python -m unittest backends/vulkan/test/test_vulkan_delegate.py -k "*torchao*" python -m unittest backends/vulkan/test/test_vulkan_graph_builder.py + python -m unittest backends/vulkan/test/test_vulkan_dynamic.py test-coreml-bc-macos: needs: [changed-files, run-decision] diff --git a/.github/workflows/vulkan.yml b/.github/workflows/vulkan.yml index dd3de1f33fb..e1a5de7ea4a 100644 --- a/.github/workflows/vulkan.yml +++ b/.github/workflows/vulkan.yml @@ -81,6 +81,8 @@ jobs: # the pt2e/torchao e2e tests below execute on the GPU (default is OFF). CMAKE_ARGS="-DEXECUTORCH_BUILD_VULKAN=ON" PYTHON_EXECUTABLE=python ./install_executorch.sh + python -m unittest backends/vulkan/test/test_vulkan_dynamic.py + # Model coverage (mirrors test-vulkan-models-linux, on real hardware). PYTHON_EXECUTABLE=python bash backends/vulkan/test/scripts/test_model.sh --build diff --git a/backends/vulkan/_passes/squeeze_unsqueeze_inputs.py b/backends/vulkan/_passes/squeeze_unsqueeze_inputs.py index 25b28ce3117..b42490b59ab 100644 --- a/backends/vulkan/_passes/squeeze_unsqueeze_inputs.py +++ b/backends/vulkan/_passes/squeeze_unsqueeze_inputs.py @@ -72,7 +72,7 @@ def _squeezable(shape: List[int]) -> bool: squeeze_out = super().call_operator( exir_ops.edge.aten.view_copy.default, (args[0], squeeze_shape), - kwargs, + {}, meta, ) # call linear on squeezed output @@ -88,6 +88,6 @@ def _squeezable(shape: List[int]) -> bool: return super().call_operator( exir_ops.edge.aten.view_copy.default, (linear_out, unsqueeze_shape), - kwargs, + {}, meta, ) diff --git a/backends/vulkan/runtime/graph/ops/glsl/activations.h b/backends/vulkan/runtime/graph/ops/glsl/activations.h index 2ba0ccc467d..b4778f3a639 100644 --- a/backends/vulkan/runtime/graph/ops/glsl/activations.h +++ b/backends/vulkan/runtime/graph/ops/glsl/activations.h @@ -6,6 +6,25 @@ * LICENSE file in the root directory of this source tree. */ +float gelu_erf(float x) { + // Abramowitz and Stegun 7.1.26: maximum absolute erf error is 1.5e-7. + const float a = abs(x) * 0.7071067811865475; + const float t = 1.0 / (1.0 + 0.3275911 * a); + const float polynomial = + (((((1.061405429 * t - 1.453152027) * t) + 1.421413741) * t - + 0.284496736) * + t + + 0.254829592) * + t; + const float erf = sign(x) * (1.0 - polynomial * exp(-a * a)); + return 0.5 * x * (1.0 + erf); +} + +vec4 gelu_erf(vec4 tex) { + return vec4( + gelu_erf(tex.x), gelu_erf(tex.y), gelu_erf(tex.z), gelu_erf(tex.w)); +} + float hardswish(float x) { if (x <= -3) { return 0; diff --git a/backends/vulkan/runtime/graph/ops/glsl/unary_op.yaml b/backends/vulkan/runtime/graph/ops/glsl/unary_op.yaml index fc70b54076b..ec7475cde63 100644 --- a/backends/vulkan/runtime/graph/ops/glsl/unary_op.yaml +++ b/backends/vulkan/runtime/graph/ops/glsl/unary_op.yaml @@ -32,6 +32,8 @@ unary_op: OPERATOR: exp(X) - NAME: gelu OPERATOR: 0.5 * X * (1 + tanh(clamp(sqrt(2 / 3.141593) * (X + 0.044715 * X * X * X), -15.0, 15.0))) + - NAME: gelu_erf + OPERATOR: gelu_erf(X) - NAME: neg OPERATOR: -X - NAME: sigmoid diff --git a/backends/vulkan/runtime/graph/ops/impl/UnaryOp.cpp b/backends/vulkan/runtime/graph/ops/impl/UnaryOp.cpp index d17f57774f7..d17979a4fa2 100644 --- a/backends/vulkan/runtime/graph/ops/impl/UnaryOp.cpp +++ b/backends/vulkan/runtime/graph/ops/impl/UnaryOp.cpp @@ -177,11 +177,15 @@ float get_val_or_inf(ComputeGraph& graph, const ValueRef& val, bool max) { } void gelu(ComputeGraph& graph, const std::vector& args) { - // args[1] is the `approximate` string - // https://fburl.com/code/9omngmyo - // currently only `approximate = "tanh"` is supported + const std::string approximate = graph.extract_string(args[1]); + VK_CHECK_COND(approximate == "none" || approximate == "tanh"); return add_unary_op_node( - graph, args[0], kDummyFloat, kDummyFloat, args[2], "gelu"); + graph, + args[0], + kDummyFloat, + kDummyFloat, + args[2], + approximate == "tanh" ? "gelu" : "gelu_erf"); } DEFINE_ACTIVATION_FN(abs); diff --git a/backends/vulkan/test/op_tests/cases.py b/backends/vulkan/test/op_tests/cases.py index dc551e8ff5e..ea98b1389a1 100644 --- a/backends/vulkan/test/op_tests/cases.py +++ b/backends/vulkan/test/op_tests/cases.py @@ -2006,12 +2006,13 @@ def get_native_batch_norm_inputs(): def get_gelu_inputs(): test_suite = VkTestSuite( [ - ((M1), "tanh"), - ((M1, M2), "tanh"), - ((S1, M1, M2), "tanh"), - ((S1, S2, S2, M2), "tanh"), + (shape, approximate) + for shape in ((M1,), (M1, M2), (S1, M1, M2), (S1, S2, S2, M2)) + for approximate in ("none", "tanh") ] ) + test_suite.data_range = (-6, 6) + test_suite.storage_types = ["utils::kTexture3D", "utils::kBuffer"] return test_suite diff --git a/backends/vulkan/test/targets.bzl b/backends/vulkan/test/targets.bzl index 5734e733195..d18e508341d 100644 --- a/backends/vulkan/test/targets.bzl +++ b/backends/vulkan/test/targets.bzl @@ -29,6 +29,24 @@ def define_common_targets(is_fbcode = False): ], ) + python_unittest( + name = "test_vulkan_dynamic", + srcs = ["test_vulkan_dynamic.py"], + env = {"ETVK_USING_SWIFTSHADER": "1"}, + preload_deps = [ + "fbsource//third-party/swiftshader/lib/linux-x64:libvk_swiftshader_fbcode", + "//executorch/backends/vulkan:vulkan_backend_lib", + "//executorch/kernels/portable:custom_ops_generated_lib", + ], + deps = [ + "//caffe2:torch", + "//executorch/backends/vulkan/partitioner:vulkan_partitioner", + "//executorch/backends/vulkan/serialization:lib", + "//executorch/exir:lib", + "//executorch/extension/pybindings:portable_lib", # @manual + ], + ) + python_unittest( name = "test_vulkan_graph_builder", srcs = ["test_vulkan_graph_builder.py"], diff --git a/backends/vulkan/test/test_vulkan_dynamic.py b/backends/vulkan/test/test_vulkan_dynamic.py new file mode 100644 index 00000000000..b671337d2a0 --- /dev/null +++ b/backends/vulkan/test/test_vulkan_dynamic.py @@ -0,0 +1,166 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + + +import math + +import operator + +import os + +import unittest + +import torch + +from executorch.backends.vulkan.partitioner.vulkan_partitioner import VulkanPartitioner + +from executorch.backends.vulkan.serialization.vulkan_graph_schema import ( + VkDataType, + VkStorageType, + VkTensor, +) + +from executorch.backends.vulkan.serialization.vulkan_graph_serialize import ( + extract_vk_flatbuffer, + flatbuffer_to_vk_graph, +) + +from executorch.exir import EdgeCompileConfig, to_edge_transform_and_lower + +from executorch.exir.lowered_backend_module import LoweredBackendModule + +from torch.export import Dim, export + +USING_SWIFTSHADER = os.environ.get("ETVK_USING_SWIFTSHADER") in ("1", "True") + + +def _vulkan_graphs(edge): + return [ + flatbuffer_to_vk_graph(extract_vk_flatbuffer(module.processed_bytes)) + for module in edge.exported_program().graph_module.modules() + if isinstance(module, LoweredBackendModule) + and module.backend_id == "VulkanBackend" + ] + + +class TestVulkanDynamic(unittest.TestCase): + def _lower( + self, + model, + inputs, + dynamic_shapes=None, + storage=VkStorageType.TEXTURE_3D, + *, + fully_delegated=True, + downcast_64_bit=True, + ): + options = { + "require_dynamic_shapes": True, + "storage_type_override": storage, + "downcast_64_bit": downcast_64_bit, + } + if storage == VkStorageType.BUFFER: + options["texture_limits"] = (1, 1, 1) + edge = to_edge_transform_and_lower( + export(model.eval(), inputs, dynamic_shapes=dynamic_shapes, strict=False), + partitioner=[VulkanPartitioner(options)], + compile_config=EdgeCompileConfig(_check_ir_validity=False), + ) + targets = [ + node.target + for node in edge.exported_program().graph.nodes + if node.op == "call_function" and node.target != operator.getitem + ] + if fully_delegated: + self.assertEqual(targets, [torch.ops.higher_order.executorch_call_delegate]) + if storage == VkStorageType.BUFFER: + for graph in _vulkan_graphs(edge): + for value_id in graph.input_ids + graph.output_ids: + value = graph.values[value_id].value + if isinstance(value, VkTensor) and math.prod(value.dims) > 4: + self.assertEqual(value.storage_type, VkStorageType.BUFFER) + return edge + + def _run( + self, + edge, + model, + inputs, + *, + atol=1e-5, + rtol=1e-4, + equal_nan=False, + check_signed_zero=False, + ): + from executorch.extension.pybindings.portable_lib import ( + _load_for_executorch_from_buffer, + ) + + if USING_SWIFTSHADER and any( + isinstance(value.value, VkTensor) + and value.value.constant_id < 0 + and value.value.datatype == VkDataType.BOOL + and value.value.storage_type == VkStorageType.BUFFER + for graph in _vulkan_graphs(edge) + for value in graph.values + ): + self.skipTest("SwiftShader does not support 8-bit storage buffers") + + program_buffer = edge.to_executorch().buffer + module = _load_for_executorch_from_buffer(program_buffer) + for sample in inputs: + with self.subTest(shapes=[tuple(x.shape) for x in sample]): + actual = module.run_method("forward", sample) + expected = model(*sample) + if isinstance(expected, torch.Tensor): + expected = (expected,) + self.assertEqual(len(actual), len(expected)) + for output, reference in zip(actual, expected): + torch.testing.assert_close( + output, reference, atol=atol, rtol=rtol, equal_nan=equal_nan + ) + if check_signed_zero: + zeros = reference == 0 + self.assertTrue( + torch.equal( + torch.signbit(output[zeros]), + torch.signbit(reference[zeros]), + ) + ) + + def test_dynamic_gelu(self): + for approximate in ("none", "tanh"): + for storage in (VkStorageType.TEXTURE_3D, VkStorageType.BUFFER): + for dtype in (torch.float32, torch.float16): + with self.subTest( + approximate=approximate, storage=storage, dtype=dtype + ): + model = torch.nn.GELU(approximate=approximate) + inputs = [ + (torch.linspace(-6, 6, s, dtype=dtype).repeat(3, 1),) + for s in (257, 17, 511, 2, 257) + ] + edge = self._lower( + model, inputs[0], ({1: Dim("s", min=2, max=512)},), storage + ) + tolerance = 5e-6 if dtype == torch.float32 else 1e-3 + self._run(edge, model, inputs, atol=tolerance, rtol=tolerance) + + def test_gelu_with_singleton_dimensions(self): + for approximate in ("none", "tanh"): + for shape in ((6, 1, 3), (2, 1, 3, 5)): + for storage in (VkStorageType.TEXTURE_3D, VkStorageType.BUFFER): + with self.subTest( + approximate=approximate, shape=shape, storage=storage + ): + model = torch.nn.GELU(approximate=approximate) + x = torch.linspace(-6, 6, math.prod(shape)).reshape(shape) + edge = self._lower(model, (x,), storage=storage) + self._run(edge, model, [(x,)], atol=5e-6, rtol=5e-6) + + +if __name__ == "__main__": + unittest.main() From fb905cfe16ae971a772901f2baeba4a4a7f3b08c Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Tue, 29 Sep 2026 11:02:05 -0400 Subject: [PATCH 5/8] [INITIAL] Recreate the Vulkan transformer and conformance stack with ghstack [ghstack-poisoned] --- .../runtime/graph/ops/glsl/convert.glslh | 41 ++++++++--- .../vulkan/runtime/graph/ops/glsl/reduce.glsl | 22 +++++- .../vulkan/runtime/graph/ops/glsl/reduce.yaml | 4 +- .../graph/ops/glsl/reduce_op_defs.glslh | 17 +++-- backends/vulkan/test/test_vulkan_dynamic.py | 72 +++++++++++++++++++ 5 files changed, 140 insertions(+), 16 deletions(-) diff --git a/backends/vulkan/runtime/graph/ops/glsl/convert.glslh b/backends/vulkan/runtime/graph/ops/glsl/convert.glslh index b901bc7e9d9..2c3c441b904 100644 --- a/backends/vulkan/runtime/graph/ops/glsl/convert.glslh +++ b/backends/vulkan/runtime/graph/ops/glsl/convert.glslh @@ -13,16 +13,41 @@ #ifdef T -#if T == float16_t - -#define convert_to_T(x) T(clamp(x, -65504, 65504)); - -#else - #define convert_to_T(x) T(x); -#endif // T == float16_t - #endif // T +float round_to_half_rte(float value) { + const uint bits = floatBitsToUint(value); + const uint exponent = (bits >> 23u) & 0xffu; + if (exponent == 0xffu) { + return value; + } + + uint result = 0u; + if (exponent >= 143u) { + result = 0x7c00u; + } else if (exponent >= 102u) { + const bool normal = exponent >= 113u; + const uint significand = (bits & 0x7fffffu) | (normal ? 0u : 0x800000u); + const uint shift = normal ? 13u : 126u - exponent; + result = (normal ? (exponent - 112u) << 10u : 0u) + (significand >> shift); + const uint remainder = significand & ((1u << shift) - 1u); + const uint halfway = 1u << (shift - 1u); + result += uint(remainder > halfway || (remainder == halfway && (result & 1u) != 0u)); + } + + // packHalf2x16 does not guarantee round-to-nearest-even on every driver. + result |= (bits >> 16u) & 0x8000u; + return unpackHalf2x16(result).x; +} + +vec4 round_to_half_rte(vec4 value) { + return vec4( + round_to_half_rte(value.x), + round_to_half_rte(value.y), + round_to_half_rte(value.z), + round_to_half_rte(value.w)); +} + #endif // CONVERT_GLSLH diff --git a/backends/vulkan/runtime/graph/ops/glsl/reduce.glsl b/backends/vulkan/runtime/graph/ops/glsl/reduce.glsl index 029e3b16756..f37a0fc773a 100644 --- a/backends/vulkan/runtime/graph/ops/glsl/reduce.glsl +++ b/backends/vulkan/runtime/graph/ops/glsl/reduce.glsl @@ -49,6 +49,7 @@ shared vec4 shared_vecs[MAX_NTHREADS]; #include "indexing_utils.h" #include "indexing.glslh" +#include "convert.glslh" int tid_to_smi(const ivec2 tid) { return tid.x + tid.y * NWORKERS; @@ -84,7 +85,26 @@ int tid_to_smi(const ivec2 tid) { #define UPDATE_ACCUM(accum, new_val) ${UPDATE_ACCUM} // Useful for operators such as mean which want to perform a final calculation // with the accumulator. -#define POSTPROCESS(accum) ${POSTPROCESS} +$if DTYPE == "half": + #define POSTPROCESS(accum) round_to_half_rte(${POSTPROCESS}) +$else: + #define POSTPROCESS(accum) ${POSTPROCESS} + +float max_propagate_nan(float a, float b) { + return isnan(a) ? a : (isnan(b) ? b : max(a, b)); +} + +vec4 max_propagate_nan(vec4 a, vec4 b) { + return mix(mix(max(a, b), b, isnan(b)), a, isnan(a)); +} + +float min_propagate_nan(float a, float b) { + return isnan(a) ? a : (isnan(b) ? b : min(a, b)); +} + +vec4 min_propagate_nan(vec4 a, vec4 b) { + return mix(mix(min(a, b), b, isnan(b)), a, isnan(a)); +} /* * Computes reduction where the reduction dim is orthogonal to the packed dim. diff --git a/backends/vulkan/runtime/graph/ops/glsl/reduce.yaml b/backends/vulkan/runtime/graph/ops/glsl/reduce.yaml index 21a7132b8db..f85d8405b2d 100644 --- a/backends/vulkan/runtime/graph/ops/glsl/reduce.yaml +++ b/backends/vulkan/runtime/graph/ops/glsl/reduce.yaml @@ -21,9 +21,9 @@ reduce: POSTPROCESS: (accum / tin_sizes[reduce_dim]) - NAME: amax INIT_ACCUM: first_val - UPDATE_ACCUM: max(accum, new_val) + UPDATE_ACCUM: max_propagate_nan(accum, new_val) POSTPROCESS: accum - NAME: amin INIT_ACCUM: first_val - UPDATE_ACCUM: min(accum, new_val) + UPDATE_ACCUM: min_propagate_nan(accum, new_val) POSTPROCESS: accum diff --git a/backends/vulkan/runtime/graph/ops/glsl/reduce_op_defs.glslh b/backends/vulkan/runtime/graph/ops/glsl/reduce_op_defs.glslh index e5f61da7586..338ad955760 100644 --- a/backends/vulkan/runtime/graph/ops/glsl/reduce_op_defs.glslh +++ b/backends/vulkan/runtime/graph/ops/glsl/reduce_op_defs.glslh @@ -15,6 +15,11 @@ struct Accum { uint count; }; +bool is_earlier_nan(ACCUM_T val, uint idx, const Accum accum) { + return isnan(float(val)) && + (!isnan(float(accum.val)) || idx < accum.idx); +} + void init_accum(out Accum accum, T val, uint idx) { accum.val = ACCUM_T(val); accum.idx = idx; @@ -40,14 +45,14 @@ void merge_accum_sum(inout Accum accum, const Accum other) { } void postprocess_accum_mean(inout Accum accum) { - accum.val /= T(accum.count); + accum.val /= ACCUM_T(accum.count); } // Amax (maximum value) void update_accum_amax(inout Accum accum, T in_val, uint idx) { ACCUM_T val = ACCUM_T(in_val); - if (val > accum.val) { + if (val > accum.val || is_earlier_nan(val, idx, accum)) { accum.val = val; accum.idx = idx; } @@ -58,7 +63,7 @@ void update_accum_amax(inout Accum accum, T in_val, uint idx) { } void merge_accum_amax(inout Accum accum, const Accum other) { - if (other.val > accum.val) { + if (other.val > accum.val || is_earlier_nan(other.val, other.idx, accum)) { accum.val = other.val; accum.idx = other.idx; } @@ -72,7 +77,7 @@ void merge_accum_amax(inout Accum accum, const Accum other) { void update_accum_amin(inout Accum accum, T in_val, uint idx) { ACCUM_T val = ACCUM_T(in_val); - if (val < accum.val) { + if (val < accum.val || is_earlier_nan(val, idx, accum)) { accum.val = val; accum.idx = idx; } @@ -83,7 +88,9 @@ void update_accum_amin(inout Accum accum, T in_val, uint idx) { } void merge_accum_amin(inout Accum accum, const Accum other) { - if (other.count > 0 && (accum.count == 0 || other.val < accum.val)) { + if (other.count > 0 && + (accum.count == 0 || other.val < accum.val || + is_earlier_nan(other.val, other.idx, accum))) { accum.val = other.val; accum.idx = other.idx; } diff --git a/backends/vulkan/test/test_vulkan_dynamic.py b/backends/vulkan/test/test_vulkan_dynamic.py index b671337d2a0..089c14eae4b 100644 --- a/backends/vulkan/test/test_vulkan_dynamic.py +++ b/backends/vulkan/test/test_vulkan_dynamic.py @@ -161,6 +161,78 @@ def test_gelu_with_singleton_dimensions(self): edge = self._lower(model, (x,), storage=storage) self._run(edge, model, [(x,)], atol=5e-6, rtol=5e-6) + def test_buffer_reduction_range(self): + class Reduce(torch.nn.Module): + def __init__(self, op): + super().__init__() + self.op = op + + def forward(self, x): + return self.op(x, dim=-1, keepdim=True) + + for op in (torch.sum, torch.mean, torch.amax): + with self.subTest(op=op): + width, value = (20000, 4) if op == torch.sum else (8, 80000) + x = torch.tensor([value, -value], dtype=torch.float32)[:, None].repeat( + 1, width + ) + model = Reduce(op) + edge = self._lower(model, (x,), storage=VkStorageType.BUFFER) + self._run(edge, model, [(x,)], atol=0, rtol=0) + + def test_reduction_special_values(self): + class Reduce(torch.nn.Module): + def __init__(self, op): + super().__init__() + self.op = op + + def forward(self, x): + return self.op(x, dim=-1, keepdim=True) + + for dtype in (torch.float32, torch.float16): + x = torch.tensor( + [ + [40000] * 9, + [-40000] * 9, + [1, torch.nan, 2, 3, torch.nan, 4, 5, 6, 7], + ], + dtype=dtype, + ) + for op in (torch.sum, torch.mean, torch.amax, torch.amin): + for storage in (VkStorageType.TEXTURE_3D, VkStorageType.BUFFER): + with self.subTest(dtype=dtype, op=op, storage=storage): + model = Reduce(op) + edge = self._lower(model, (x,), storage=storage) + self._run(edge, model, [(x,)], atol=0, rtol=0, equal_nan=True) + + x = torch.tensor([4, -4], dtype=torch.float16)[:, None].repeat(1, 70000) + model = Reduce(torch.mean) + edge = self._lower(model, (x,), storage=VkStorageType.BUFFER) + self._run(edge, model, [(x,)], atol=0, rtol=0) + + def test_argreduce_first_nan(self): + class Reduce(torch.nn.Module): + def __init__(self, op): + super().__init__() + self.op = op + + def forward(self, x): + return self.op(x, dim=-1, keepdim=True) + + for dtype in (torch.float32, torch.float16): + x = torch.tensor( + [ + [1, torch.nan, 2, torch.nan, 3, 4, 5], + [1, 2, 3, 4, 5, torch.nan, torch.nan], + ], + dtype=dtype, + ) + for op in (torch.argmax, torch.argmin): + with self.subTest(dtype=dtype, op=op): + model = Reduce(op) + edge = self._lower(model, (x,), storage=VkStorageType.BUFFER) + self._run(edge, model, [(x,)], atol=0, rtol=0) + if __name__ == "__main__": unittest.main() From f963896d7b6647bcf9cf352ee57846ec6238c3cc Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Fri, 2 Oct 2026 22:54:36 -0400 Subject: [PATCH 6/8] [UPDATE] Cover exact FP16 halfway rounding in texture reductions [ghstack-poisoned] --- backends/vulkan/test/test_vulkan_dynamic.py | 25 +++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/backends/vulkan/test/test_vulkan_dynamic.py b/backends/vulkan/test/test_vulkan_dynamic.py index 089c14eae4b..0038e3fca27 100644 --- a/backends/vulkan/test/test_vulkan_dynamic.py +++ b/backends/vulkan/test/test_vulkan_dynamic.py @@ -210,6 +210,31 @@ def forward(self, x): edge = self._lower(model, (x,), storage=VkStorageType.BUFFER) self._run(edge, model, [(x,)], atol=0, rtol=0) + def test_fp16_reduction_halfway_rounding(self): + class Mean(torch.nn.Module): + def forward(self, x): + return torch.mean(x, dim=-1, keepdim=True) + + # Adjacent half values produce ties at even/odd mantissas, an exponent + # carry, and the normal/subnormal boundary without rounding the inputs. + x = torch.tensor( + [ + [1, 1 + 2**-10], + [1 + 2**-10, 1 + 2**-9], + [2 - 2**-10, 2], + [2**-14 - 2**-24, 2**-14], + ], + dtype=torch.float16, + ) + x = torch.cat((x, -x)) + model = Mean() + edge = self._lower(model, (x,), storage=VkStorageType.TEXTURE_3D) + (graph,) = _vulkan_graphs(edge) + output = graph.values[graph.output_ids[0]].value + self.assertEqual(output.datatype, VkDataType.FLOAT16) + self.assertEqual(output.storage_type, VkStorageType.TEXTURE_3D) + self._run(edge, model, [(x,)], atol=0, rtol=0) + def test_argreduce_first_nan(self): class Reduce(torch.nn.Module): def __init__(self, op): From 7baf82c0358b09a1a9e5b260fcf16e92802b5c3f Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Fri, 2 Oct 2026 23:46:29 -0400 Subject: [PATCH 7/8] [UPDATE] Add exact FP16 zero/subnormal and overflow-threshold coverage [ghstack-poisoned] --- backends/vulkan/test/test_vulkan_dynamic.py | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/backends/vulkan/test/test_vulkan_dynamic.py b/backends/vulkan/test/test_vulkan_dynamic.py index 0038e3fca27..9ad43128044 100644 --- a/backends/vulkan/test/test_vulkan_dynamic.py +++ b/backends/vulkan/test/test_vulkan_dynamic.py @@ -223,6 +223,8 @@ def forward(self, x): [1 + 2**-10, 1 + 2**-9], [2 - 2**-10, 2], [2**-14 - 2**-24, 2**-14], + [0, 2**-24], + [2**-24, 2**-23], ], dtype=torch.float16, ) @@ -233,6 +235,21 @@ def forward(self, x): output = graph.values[graph.output_ids[0]].value self.assertEqual(output.datatype, VkDataType.FLOAT16) self.assertEqual(output.storage_type, VkStorageType.TEXTURE_3D) + self._run(edge, model, [(x,)], atol=0, rtol=0, check_signed_zero=True) + + def test_fp16_reduction_overflow_rounding(self): + class Sum(torch.nn.Module): + def forward(self, x): + return torch.sum(x, dim=-1, keepdim=True) + + x = torch.tensor([[65504, 15], [65504, 16], [65504, 17]], dtype=torch.float16) + x = torch.cat((x, -x)) + model = Sum() + edge = self._lower(model, (x,), storage=VkStorageType.TEXTURE_3D) + (graph,) = _vulkan_graphs(edge) + output = graph.values[graph.output_ids[0]].value + self.assertEqual(output.datatype, VkDataType.FLOAT16) + self.assertEqual(output.storage_type, VkStorageType.TEXTURE_3D) self._run(edge, model, [(x,)], atol=0, rtol=0) def test_argreduce_first_nan(self): From 1ad69df92cfe11c70e4d948512dcac4958ab6a25 Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Fri, 2 Oct 2026 23:58:47 -0400 Subject: [PATCH 8/8] [UPDATE] Exercise int32 reduction shaders directly beyond the removed clamp [ghstack-poisoned] --- backends/vulkan/test/test_vulkan_dynamic.py | 53 ++++++++++++++++++++- 1 file changed, 51 insertions(+), 2 deletions(-) diff --git a/backends/vulkan/test/test_vulkan_dynamic.py b/backends/vulkan/test/test_vulkan_dynamic.py index 9ad43128044..f911d7ca22d 100644 --- a/backends/vulkan/test/test_vulkan_dynamic.py +++ b/backends/vulkan/test/test_vulkan_dynamic.py @@ -15,7 +15,10 @@ import torch -from executorch.backends.vulkan.partitioner.vulkan_partitioner import VulkanPartitioner +from executorch.backends.vulkan.partitioner.vulkan_partitioner import ( + parse_compile_options, + VulkanPartitioner, +) from executorch.backends.vulkan.serialization.vulkan_graph_schema import ( VkDataType, @@ -28,7 +31,9 @@ flatbuffer_to_vk_graph, ) -from executorch.exir import EdgeCompileConfig, to_edge_transform_and_lower +from executorch.exir import EdgeCompileConfig, to_edge, to_edge_transform_and_lower + +from executorch.exir.backend.backend_api import to_backend from executorch.exir.lowered_backend_module import LoweredBackendModule @@ -180,6 +185,50 @@ def forward(self, x): edge = self._lower(model, (x,), storage=VkStorageType.BUFFER) self._run(edge, model, [(x,)], atol=0, rtol=0) + def test_int32_buffer_reduction_shader_range(self): + from executorch.extension.pybindings.portable_lib import ( + _load_for_executorch_from_buffer, + ) + + class Reduce(torch.nn.Module): + def __init__(self, op): + super().__init__() + self.op = op + + def forward(self, x): + return self.op(x, dim=-1, keepdim=True) + + x = torch.tensor( + [[80000, 80001, 80002, 80003], [-80000, -80001, -80002, -80003]], + dtype=torch.int32, + ) + for op in (torch.amax, torch.amin): + with self.subTest(op=op): + model = Reduce(op) + edge = to_edge(export(model, (x,))) + # Integer reductions are excluded by the partitioner. + lowered = to_backend( + "VulkanBackend", + edge.exported_program(), + parse_compile_options( + { + "storage_type_override": VkStorageType.BUFFER, + "texture_limits": (1, 1, 1), + } + ), + ) + graph = flatbuffer_to_vk_graph( + extract_vk_flatbuffer(lowered.processed_bytes) + ) + for value_id in graph.input_ids + graph.output_ids: + value = graph.values[value_id].value + self.assertEqual(value.datatype, VkDataType.INT32) + self.assertEqual(value.storage_type, VkStorageType.BUFFER) + program_buffer = lowered.buffer() + module = _load_for_executorch_from_buffer(program_buffer) + (actual,) = module.run_method("forward", (x,)) + torch.testing.assert_close(actual, model(x), atol=0, rtol=0) + def test_reduction_special_values(self): class Reduce(torch.nn.Module): def __init__(self, op):