diff --git a/.ci/scripts/setup-vulkan-linux-deps.sh b/.ci/scripts/setup-vulkan-linux-deps.sh index debe610a18a..c1981d5050b 100755 --- a/.ci/scripts/setup-vulkan-linux-deps.sh +++ b/.ci/scripts/setup-vulkan-linux-deps.sh @@ -67,35 +67,37 @@ install_vulkan_loader() { # libvulkan.so.1 (the Khronos loader that volk dlopen()s at runtime) is not part # of the NVIDIA driver and is absent from the CUDA builder image; vulkan-tools # provides vulkaninfo for the device sanity check. Both ship as native el8 RPMs. + # The NVIDIA ICD also needs EGL and X11 client libraries on headless runners. if command -v dnf >/dev/null 2>&1; then - _maybe_sudo dnf install -y vulkan-loader vulkan-tools + _maybe_sudo dnf install -y vulkan-loader vulkan-tools libglvnd-egl libX11 libXext fi } _find_nvidia_vulkan_library() { - # NVIDIA implements its Vulkan ICD inside libGLX_nvidia.so.0. The NVIDIA - # container runtime mounts this library into the container (it is pulled from - # the driver's ldcache when NVIDIA_DRIVER_CAPABILITIES includes graphics/all), - # so prefer ldconfig and fall back to the usual mount locations. - local lib cand - lib="$(ldconfig -p 2>/dev/null | awk '/libGLX_nvidia\.so\.0/ {print $NF; exit}')" - if [ -z "${lib}" ]; then - for cand in /usr/lib64/libGLX_nvidia.so.0 \ - /usr/lib/x86_64-linux-gnu/libGLX_nvidia.so.0 \ - /usr/lib/libGLX_nvidia.so.0; do + # NVIDIA provides EGL and GLX Vulkan ICDs. Prefer EGL on headless runners; + # the GLX entry point can fail to initialize without an X server. + local soname lib cand + for soname in libEGL_nvidia.so.0 libGLX_nvidia.so.0; do + lib="$(ldconfig -p 2>/dev/null | awk -v name="${soname}" '$1 == name {print $NF; exit}')" + if [ -n "${lib}" ]; then + printf '%s' "${lib}" + return + fi + for cand in "/usr/lib64/${soname}" \ + "/usr/lib/x86_64-linux-gnu/${soname}" \ + "/usr/lib/${soname}"; do if [ -e "${cand}" ]; then - lib="${cand}" - break + printf '%s' "${cand}" + return fi done - fi - printf '%s' "${lib}" + done } _vulkan_has_real_device() { # True if the loader enumerates a hardware GPU. vulkaninfo can exit non-zero # for unrelated reasons (no display/WSI), so key off the reported deviceType. - command -v vulkaninfo >/dev/null 2>&1 || return 0 + command -v vulkaninfo >/dev/null 2>&1 || return 1 vulkaninfo --summary 2>/dev/null | grep -qE 'PHYSICAL_DEVICE_TYPE_(DISCRETE|INTEGRATED|VIRTUAL)_GPU' } @@ -131,25 +133,24 @@ JSON echo "Real NVIDIA GPU selected; pinned Vulkan ICD to ${nvidia_lib}" return fi - echo "WARNING: ${nvidia_lib} present but no GPU enumerated; using SwiftShader." - # Surface why the NVIDIA driver did not enumerate (e.g. a missing dependency - # of libGLX_nvidia, or no render node) so the fallback is diagnosable in CI. + echo "ERROR: ${nvidia_lib} present but no GPU enumerated." + # Surface missing ICD dependencies and device-enumeration errors. + ldd "${nvidia_lib}" || true if command -v vulkaninfo >/dev/null 2>&1; then echo "--- NVIDIA Vulkan ICD diagnostic ---" VK_LOADER_DEBUG=warn vulkaninfo --summary 2>&1 | head -40 || true echo "--- end diagnostic ---" fi - unset VK_ICD_FILENAMES else - echo "WARNING: no NVIDIA Vulkan driver library found; using SwiftShader." + echo "ERROR: no NVIDIA Vulkan driver library found." fi - install_swiftshader + return 1 } VULKAN_SDK_VERSION="1.4.321.1" # The no-argument default installs SwiftShader so the existing CPU-runner CI is -# unchanged. Pass "real-gpu" to prefer a real system ICD when one is present. +# unchanged. Pass "real-gpu" to require a real system ICD. case "${1:-swiftshader}" in real-gpu) # Do not download the LunarG SDK here: its prebuilt glslc cannot run on the diff --git a/.ci/scripts/test_backend.sh b/.ci/scripts/test_backend.sh index 068d5adb260..8bfe33333a8 100755 --- a/.ci/scripts/test_backend.sh +++ b/.ci/scripts/test_backend.sh @@ -54,15 +54,8 @@ if [[ "$FLOW" == *qnn* ]]; then fi if [[ "$FLOW" == *vulkan* ]]; then - # Setup the Vulkan SDK and select an ICD: use the real system GPU ICD when one - # is present (real-GPU runner), otherwise fall back to SwiftShader (CPU - # runner). The Vulkan loader searches both standard ICD directories. - if ls /etc/vulkan/icd.d/*.json /usr/share/vulkan/icd.d/*.json \ - >/dev/null 2>&1; then - source .ci/scripts/setup-vulkan-linux-deps.sh "real-gpu" - else - source .ci/scripts/setup-vulkan-linux-deps.sh "swiftshader" - fi + # CPU runners can have Mesa ICDs installed without a usable hardware GPU. + source .ci/scripts/setup-vulkan-linux-deps.sh "swiftshader" EXTRA_BUILD_ARGS+=" -DEXECUTORCH_BUILD_VULKAN=ON" fi diff --git a/.github/workflows/pull.yml b/.github/workflows/pull.yml index 2f91799b9a4..a70a824f223 100644 --- a/.github/workflows/pull.yml +++ b/.github/workflows/pull.yml @@ -1682,6 +1682,7 @@ jobs: # route in the future. python -m unittest backends/vulkan/test/test_vulkan_delegate.py -k "*pt2e*" python -m unittest backends/vulkan/test/test_vulkan_delegate.py -k "*torchao*" + python -m unittest backends/vulkan/test/test_vulkan_graph_builder.py test-coreml-bc-macos: needs: [changed-files, run-decision] diff --git a/.github/workflows/vulkan.yml b/.github/workflows/vulkan.yml index 2514a8d582c..dd3de1f33fb 100644 --- a/.github/workflows/vulkan.yml +++ b/.github/workflows/vulkan.yml @@ -70,10 +70,7 @@ jobs: # not run, so setup-vulkan-linux-deps.sh sources those from conda-forge and # the system package manager instead. The NVIDIA container runtime mounts # the driver's Vulkan library but not its ICD manifest, so the script - # synthesizes one and pins the loader to it; if no NVIDIA library is found - # it falls back to SwiftShader. - # NOTE: first-run check - inspect the vulkaninfo output below to confirm a - # real NVIDIA device is selected (not llvmpipe/SwiftShader). + # synthesizes one and requires the loader to enumerate a hardware GPU. source .ci/scripts/setup-vulkan-linux-deps.sh real-gpu vulkaninfo --summary || true @@ -100,23 +97,16 @@ jobs: # Operator coverage (mirrors test-vulkan-operators-linux, on real hardware). # The custom-op prototyping binaries are GPU microbenchmarks that rely on - # GPU timestamp queries; they need a real device and crash on the - # SwiftShader software fallback. Always build them (compile coverage), but - # only run them when a real GPU was selected (setup-vulkan-linux-deps.sh - # exports ETVK_USING_SWIFTSHADER when it falls back to SwiftShader). + # GPU timestamp queries, so they require the hardware selected above. PYTHON_EXECUTABLE=python bash backends/vulkan/test/custom_ops/build_and_run.sh - if [ -z "${ETVK_USING_SWIFTSHADER:-}" ]; then - ./cmake-out/backends/vulkan/test/custom_ops/test_add - ./cmake-out/backends/vulkan/test/custom_ops/test_q8csw_linear - ./cmake-out/backends/vulkan/test/custom_ops/test_q8csw_conv2d - ./cmake-out/backends/vulkan/test/custom_ops/test_q4gsw_linear - ./cmake-out/backends/vulkan/test/custom_ops/test_choose_qparams_per_row - ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_qdq - ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_clone - ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_binary - else - echo "SwiftShader fallback active: built custom-op benchmarks but skipping execution (they require real-GPU timestamp queries)." - fi + ./cmake-out/backends/vulkan/test/custom_ops/test_add + ./cmake-out/backends/vulkan/test/custom_ops/test_q8csw_linear + ./cmake-out/backends/vulkan/test/custom_ops/test_q8csw_conv2d + ./cmake-out/backends/vulkan/test/custom_ops/test_q4gsw_linear + ./cmake-out/backends/vulkan/test/custom_ops/test_choose_qparams_per_row + ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_qdq + ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_clone + ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_binary PYTHON_EXECUTABLE=python bash backends/vulkan/test/scripts/test_op.sh --build diff --git a/backends/vulkan/serialization/vulkan_graph_builder.py b/backends/vulkan/serialization/vulkan_graph_builder.py index 3de60966422..8703d391ce0 100644 --- a/backends/vulkan/serialization/vulkan_graph_builder.py +++ b/backends/vulkan/serialization/vulkan_graph_builder.py @@ -231,13 +231,7 @@ def create_null_value(self) -> int: return new_id def get_or_create_scalar_value(self, scalar: _ScalarType) -> int: - scalar_key = scalar - # Since Python considers 1 and True to be "equivalent" (as well as 0 and False) - # to distinguish entries in the dictionary, if scalar is bool then convert it - # to a string representation to use as a key for the dictionary - if isinstance(scalar, bool): - scalar_key = str(scalar) - + scalar_key = (type(scalar), repr(scalar)) if scalar_key in self.const_scalar_to_value_ids: return self.const_scalar_to_value_ids[scalar_key] diff --git a/backends/vulkan/test/targets.bzl b/backends/vulkan/test/targets.bzl index 77e74f9c8fc..5734e733195 100644 --- a/backends/vulkan/test/targets.bzl +++ b/backends/vulkan/test/targets.bzl @@ -29,6 +29,17 @@ def define_common_targets(is_fbcode = False): ], ) + python_unittest( + name = "test_vulkan_graph_builder", + srcs = ["test_vulkan_graph_builder.py"], + deps = [ + "//caffe2:torch", + "//executorch/backends/vulkan/serialization:lib", + "//executorch/backends/vulkan:vulkan_preprocess", + "//executorch/exir:lib", + ], + ) + python_unittest( name = "test_vulkan_passes", srcs = [ diff --git a/backends/vulkan/test/test_vulkan_graph_builder.py b/backends/vulkan/test/test_vulkan_graph_builder.py index 65afc3a2542..c180308e77a 100644 --- a/backends/vulkan/test/test_vulkan_graph_builder.py +++ b/backends/vulkan/test/test_vulkan_graph_builder.py @@ -8,12 +8,40 @@ import torch from executorch.backends.vulkan.serialization.vulkan_graph_builder import VkGraphBuilder +from executorch.backends.vulkan.serialization.vulkan_graph_schema import ( + Bool, + Double, + Int, +) from executorch.backends.vulkan.vulkan_preprocess import apply_passes from executorch.exir import to_edge from executorch.exir.backend.utils import DelegateMappingBuilder from executorch.exir.passes import SpecPropPass +class TestVkGraphBuilderScalarTensor(unittest.TestCase): + def test_scalar_cache_preserves_types_and_signed_zero(self): + program = torch.export.export(torch.nn.Identity(), (torch.ones(1),)) + builder = VkGraphBuilder( + program, DelegateMappingBuilder(generated_identifiers=True) + ) + scalars = (1.0, 1, True, 0.0, -0.0, 0, False) + expected = ( + Double(1.0), + Int(1), + Bool(True), + Double(0.0), + Double(-0.0), + Int(0), + Bool(False), + ) + ids = [builder.get_or_create_scalar_value(value) for value in scalars] + self.assertEqual(len(set(ids)), len(scalars)) + for value, value_id, serialized in zip(scalars, ids, expected): + self.assertEqual(builder.get_or_create_scalar_value(value), value_id) + self.assertEqual(repr(builder.values[value_id].value), repr(serialized)) + + class TestVkGraphBuilderInputIds(unittest.TestCase): """The serialized input list has to match the delegate call's arguments.