From 446a5d62fbe307f11d2d80474ce4870c10938c7a Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Tue, 29 Sep 2026 11:02:05 -0400 Subject: [PATCH] [INITIAL] Recreate the Vulkan transformer and conformance stack with ghstack [ghstack-poisoned] --- .github/workflows/pull.yml | 1 + .github/workflows/vulkan.yml | 2 + .../_passes/squeeze_unsqueeze_inputs.py | 4 +- .../runtime/graph/ops/glsl/activations.h | 19 ++ .../runtime/graph/ops/glsl/unary_op.yaml | 2 + .../vulkan/runtime/graph/ops/impl/UnaryOp.cpp | 12 +- backends/vulkan/test/op_tests/cases.py | 9 +- backends/vulkan/test/targets.bzl | 18 ++ backends/vulkan/test/test_vulkan_dynamic.py | 166 ++++++++++++++++++ 9 files changed, 223 insertions(+), 10 deletions(-) create mode 100644 backends/vulkan/test/test_vulkan_dynamic.py diff --git a/.github/workflows/pull.yml b/.github/workflows/pull.yml index a70a824f223..8c7c5108395 100644 --- a/.github/workflows/pull.yml +++ b/.github/workflows/pull.yml @@ -1683,6 +1683,7 @@ jobs: python -m unittest backends/vulkan/test/test_vulkan_delegate.py -k "*pt2e*" python -m unittest backends/vulkan/test/test_vulkan_delegate.py -k "*torchao*" python -m unittest backends/vulkan/test/test_vulkan_graph_builder.py + python -m unittest backends/vulkan/test/test_vulkan_dynamic.py test-coreml-bc-macos: needs: [changed-files, run-decision] diff --git a/.github/workflows/vulkan.yml b/.github/workflows/vulkan.yml index dd3de1f33fb..e1a5de7ea4a 100644 --- a/.github/workflows/vulkan.yml +++ b/.github/workflows/vulkan.yml @@ -81,6 +81,8 @@ jobs: # the pt2e/torchao e2e tests below execute on the GPU (default is OFF). CMAKE_ARGS="-DEXECUTORCH_BUILD_VULKAN=ON" PYTHON_EXECUTABLE=python ./install_executorch.sh + python -m unittest backends/vulkan/test/test_vulkan_dynamic.py + # Model coverage (mirrors test-vulkan-models-linux, on real hardware). PYTHON_EXECUTABLE=python bash backends/vulkan/test/scripts/test_model.sh --build diff --git a/backends/vulkan/_passes/squeeze_unsqueeze_inputs.py b/backends/vulkan/_passes/squeeze_unsqueeze_inputs.py index 25b28ce3117..b42490b59ab 100644 --- a/backends/vulkan/_passes/squeeze_unsqueeze_inputs.py +++ b/backends/vulkan/_passes/squeeze_unsqueeze_inputs.py @@ -72,7 +72,7 @@ def _squeezable(shape: List[int]) -> bool: squeeze_out = super().call_operator( exir_ops.edge.aten.view_copy.default, (args[0], squeeze_shape), - kwargs, + {}, meta, ) # call linear on squeezed output @@ -88,6 +88,6 @@ def _squeezable(shape: List[int]) -> bool: return super().call_operator( exir_ops.edge.aten.view_copy.default, (linear_out, unsqueeze_shape), - kwargs, + {}, meta, ) diff --git a/backends/vulkan/runtime/graph/ops/glsl/activations.h b/backends/vulkan/runtime/graph/ops/glsl/activations.h index 2ba0ccc467d..b4778f3a639 100644 --- a/backends/vulkan/runtime/graph/ops/glsl/activations.h +++ b/backends/vulkan/runtime/graph/ops/glsl/activations.h @@ -6,6 +6,25 @@ * LICENSE file in the root directory of this source tree. */ +float gelu_erf(float x) { + // Abramowitz and Stegun 7.1.26: maximum absolute erf error is 1.5e-7. + const float a = abs(x) * 0.7071067811865475; + const float t = 1.0 / (1.0 + 0.3275911 * a); + const float polynomial = + (((((1.061405429 * t - 1.453152027) * t) + 1.421413741) * t - + 0.284496736) * + t + + 0.254829592) * + t; + const float erf = sign(x) * (1.0 - polynomial * exp(-a * a)); + return 0.5 * x * (1.0 + erf); +} + +vec4 gelu_erf(vec4 tex) { + return vec4( + gelu_erf(tex.x), gelu_erf(tex.y), gelu_erf(tex.z), gelu_erf(tex.w)); +} + float hardswish(float x) { if (x <= -3) { return 0; diff --git a/backends/vulkan/runtime/graph/ops/glsl/unary_op.yaml b/backends/vulkan/runtime/graph/ops/glsl/unary_op.yaml index fc70b54076b..ec7475cde63 100644 --- a/backends/vulkan/runtime/graph/ops/glsl/unary_op.yaml +++ b/backends/vulkan/runtime/graph/ops/glsl/unary_op.yaml @@ -32,6 +32,8 @@ unary_op: OPERATOR: exp(X) - NAME: gelu OPERATOR: 0.5 * X * (1 + tanh(clamp(sqrt(2 / 3.141593) * (X + 0.044715 * X * X * X), -15.0, 15.0))) + - NAME: gelu_erf + OPERATOR: gelu_erf(X) - NAME: neg OPERATOR: -X - NAME: sigmoid diff --git a/backends/vulkan/runtime/graph/ops/impl/UnaryOp.cpp b/backends/vulkan/runtime/graph/ops/impl/UnaryOp.cpp index d17f57774f7..d17979a4fa2 100644 --- a/backends/vulkan/runtime/graph/ops/impl/UnaryOp.cpp +++ b/backends/vulkan/runtime/graph/ops/impl/UnaryOp.cpp @@ -177,11 +177,15 @@ float get_val_or_inf(ComputeGraph& graph, const ValueRef& val, bool max) { } void gelu(ComputeGraph& graph, const std::vector& args) { - // args[1] is the `approximate` string - // https://fburl.com/code/9omngmyo - // currently only `approximate = "tanh"` is supported + const std::string approximate = graph.extract_string(args[1]); + VK_CHECK_COND(approximate == "none" || approximate == "tanh"); return add_unary_op_node( - graph, args[0], kDummyFloat, kDummyFloat, args[2], "gelu"); + graph, + args[0], + kDummyFloat, + kDummyFloat, + args[2], + approximate == "tanh" ? "gelu" : "gelu_erf"); } DEFINE_ACTIVATION_FN(abs); diff --git a/backends/vulkan/test/op_tests/cases.py b/backends/vulkan/test/op_tests/cases.py index dc551e8ff5e..ea98b1389a1 100644 --- a/backends/vulkan/test/op_tests/cases.py +++ b/backends/vulkan/test/op_tests/cases.py @@ -2006,12 +2006,13 @@ def get_native_batch_norm_inputs(): def get_gelu_inputs(): test_suite = VkTestSuite( [ - ((M1), "tanh"), - ((M1, M2), "tanh"), - ((S1, M1, M2), "tanh"), - ((S1, S2, S2, M2), "tanh"), + (shape, approximate) + for shape in ((M1,), (M1, M2), (S1, M1, M2), (S1, S2, S2, M2)) + for approximate in ("none", "tanh") ] ) + test_suite.data_range = (-6, 6) + test_suite.storage_types = ["utils::kTexture3D", "utils::kBuffer"] return test_suite diff --git a/backends/vulkan/test/targets.bzl b/backends/vulkan/test/targets.bzl index 5734e733195..d18e508341d 100644 --- a/backends/vulkan/test/targets.bzl +++ b/backends/vulkan/test/targets.bzl @@ -29,6 +29,24 @@ def define_common_targets(is_fbcode = False): ], ) + python_unittest( + name = "test_vulkan_dynamic", + srcs = ["test_vulkan_dynamic.py"], + env = {"ETVK_USING_SWIFTSHADER": "1"}, + preload_deps = [ + "fbsource//third-party/swiftshader/lib/linux-x64:libvk_swiftshader_fbcode", + "//executorch/backends/vulkan:vulkan_backend_lib", + "//executorch/kernels/portable:custom_ops_generated_lib", + ], + deps = [ + "//caffe2:torch", + "//executorch/backends/vulkan/partitioner:vulkan_partitioner", + "//executorch/backends/vulkan/serialization:lib", + "//executorch/exir:lib", + "//executorch/extension/pybindings:portable_lib", # @manual + ], + ) + python_unittest( name = "test_vulkan_graph_builder", srcs = ["test_vulkan_graph_builder.py"], diff --git a/backends/vulkan/test/test_vulkan_dynamic.py b/backends/vulkan/test/test_vulkan_dynamic.py new file mode 100644 index 00000000000..b671337d2a0 --- /dev/null +++ b/backends/vulkan/test/test_vulkan_dynamic.py @@ -0,0 +1,166 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + + +import math + +import operator + +import os + +import unittest + +import torch + +from executorch.backends.vulkan.partitioner.vulkan_partitioner import VulkanPartitioner + +from executorch.backends.vulkan.serialization.vulkan_graph_schema import ( + VkDataType, + VkStorageType, + VkTensor, +) + +from executorch.backends.vulkan.serialization.vulkan_graph_serialize import ( + extract_vk_flatbuffer, + flatbuffer_to_vk_graph, +) + +from executorch.exir import EdgeCompileConfig, to_edge_transform_and_lower + +from executorch.exir.lowered_backend_module import LoweredBackendModule + +from torch.export import Dim, export + +USING_SWIFTSHADER = os.environ.get("ETVK_USING_SWIFTSHADER") in ("1", "True") + + +def _vulkan_graphs(edge): + return [ + flatbuffer_to_vk_graph(extract_vk_flatbuffer(module.processed_bytes)) + for module in edge.exported_program().graph_module.modules() + if isinstance(module, LoweredBackendModule) + and module.backend_id == "VulkanBackend" + ] + + +class TestVulkanDynamic(unittest.TestCase): + def _lower( + self, + model, + inputs, + dynamic_shapes=None, + storage=VkStorageType.TEXTURE_3D, + *, + fully_delegated=True, + downcast_64_bit=True, + ): + options = { + "require_dynamic_shapes": True, + "storage_type_override": storage, + "downcast_64_bit": downcast_64_bit, + } + if storage == VkStorageType.BUFFER: + options["texture_limits"] = (1, 1, 1) + edge = to_edge_transform_and_lower( + export(model.eval(), inputs, dynamic_shapes=dynamic_shapes, strict=False), + partitioner=[VulkanPartitioner(options)], + compile_config=EdgeCompileConfig(_check_ir_validity=False), + ) + targets = [ + node.target + for node in edge.exported_program().graph.nodes + if node.op == "call_function" and node.target != operator.getitem + ] + if fully_delegated: + self.assertEqual(targets, [torch.ops.higher_order.executorch_call_delegate]) + if storage == VkStorageType.BUFFER: + for graph in _vulkan_graphs(edge): + for value_id in graph.input_ids + graph.output_ids: + value = graph.values[value_id].value + if isinstance(value, VkTensor) and math.prod(value.dims) > 4: + self.assertEqual(value.storage_type, VkStorageType.BUFFER) + return edge + + def _run( + self, + edge, + model, + inputs, + *, + atol=1e-5, + rtol=1e-4, + equal_nan=False, + check_signed_zero=False, + ): + from executorch.extension.pybindings.portable_lib import ( + _load_for_executorch_from_buffer, + ) + + if USING_SWIFTSHADER and any( + isinstance(value.value, VkTensor) + and value.value.constant_id < 0 + and value.value.datatype == VkDataType.BOOL + and value.value.storage_type == VkStorageType.BUFFER + for graph in _vulkan_graphs(edge) + for value in graph.values + ): + self.skipTest("SwiftShader does not support 8-bit storage buffers") + + program_buffer = edge.to_executorch().buffer + module = _load_for_executorch_from_buffer(program_buffer) + for sample in inputs: + with self.subTest(shapes=[tuple(x.shape) for x in sample]): + actual = module.run_method("forward", sample) + expected = model(*sample) + if isinstance(expected, torch.Tensor): + expected = (expected,) + self.assertEqual(len(actual), len(expected)) + for output, reference in zip(actual, expected): + torch.testing.assert_close( + output, reference, atol=atol, rtol=rtol, equal_nan=equal_nan + ) + if check_signed_zero: + zeros = reference == 0 + self.assertTrue( + torch.equal( + torch.signbit(output[zeros]), + torch.signbit(reference[zeros]), + ) + ) + + def test_dynamic_gelu(self): + for approximate in ("none", "tanh"): + for storage in (VkStorageType.TEXTURE_3D, VkStorageType.BUFFER): + for dtype in (torch.float32, torch.float16): + with self.subTest( + approximate=approximate, storage=storage, dtype=dtype + ): + model = torch.nn.GELU(approximate=approximate) + inputs = [ + (torch.linspace(-6, 6, s, dtype=dtype).repeat(3, 1),) + for s in (257, 17, 511, 2, 257) + ] + edge = self._lower( + model, inputs[0], ({1: Dim("s", min=2, max=512)},), storage + ) + tolerance = 5e-6 if dtype == torch.float32 else 1e-3 + self._run(edge, model, inputs, atol=tolerance, rtol=tolerance) + + def test_gelu_with_singleton_dimensions(self): + for approximate in ("none", "tanh"): + for shape in ((6, 1, 3), (2, 1, 3, 5)): + for storage in (VkStorageType.TEXTURE_3D, VkStorageType.BUFFER): + with self.subTest( + approximate=approximate, shape=shape, storage=storage + ): + model = torch.nn.GELU(approximate=approximate) + x = torch.linspace(-6, 6, math.prod(shape)).reshape(shape) + edge = self._lower(model, (x,), storage=storage) + self._run(edge, model, [(x,)], atol=5e-6, rtol=5e-6) + + +if __name__ == "__main__": + unittest.main()