From aae8159bd771d52ad6822db6207655bee606820e Mon Sep 17 00:00:00 2001 From: Mergen Nachin Date: Tue, 29 Sep 2026 11:02:04 -0400 Subject: [PATCH] [INITIAL] Recreate the Vulkan transformer and conformance stack with ghstack [ghstack-poisoned] --- .ci/scripts/setup-vulkan-linux-deps.sh | 47 +++++++++++++------------- .ci/scripts/test_backend.sh | 11 ++---- .github/workflows/vulkan.yml | 30 ++++++---------- 3 files changed, 36 insertions(+), 52 deletions(-) diff --git a/.ci/scripts/setup-vulkan-linux-deps.sh b/.ci/scripts/setup-vulkan-linux-deps.sh index debe610a18a..c1981d5050b 100755 --- a/.ci/scripts/setup-vulkan-linux-deps.sh +++ b/.ci/scripts/setup-vulkan-linux-deps.sh @@ -67,35 +67,37 @@ install_vulkan_loader() { # libvulkan.so.1 (the Khronos loader that volk dlopen()s at runtime) is not part # of the NVIDIA driver and is absent from the CUDA builder image; vulkan-tools # provides vulkaninfo for the device sanity check. Both ship as native el8 RPMs. + # The NVIDIA ICD also needs EGL and X11 client libraries on headless runners. if command -v dnf >/dev/null 2>&1; then - _maybe_sudo dnf install -y vulkan-loader vulkan-tools + _maybe_sudo dnf install -y vulkan-loader vulkan-tools libglvnd-egl libX11 libXext fi } _find_nvidia_vulkan_library() { - # NVIDIA implements its Vulkan ICD inside libGLX_nvidia.so.0. The NVIDIA - # container runtime mounts this library into the container (it is pulled from - # the driver's ldcache when NVIDIA_DRIVER_CAPABILITIES includes graphics/all), - # so prefer ldconfig and fall back to the usual mount locations. - local lib cand - lib="$(ldconfig -p 2>/dev/null | awk '/libGLX_nvidia\.so\.0/ {print $NF; exit}')" - if [ -z "${lib}" ]; then - for cand in /usr/lib64/libGLX_nvidia.so.0 \ - /usr/lib/x86_64-linux-gnu/libGLX_nvidia.so.0 \ - /usr/lib/libGLX_nvidia.so.0; do + # NVIDIA provides EGL and GLX Vulkan ICDs. Prefer EGL on headless runners; + # the GLX entry point can fail to initialize without an X server. + local soname lib cand + for soname in libEGL_nvidia.so.0 libGLX_nvidia.so.0; do + lib="$(ldconfig -p 2>/dev/null | awk -v name="${soname}" '$1 == name {print $NF; exit}')" + if [ -n "${lib}" ]; then + printf '%s' "${lib}" + return + fi + for cand in "/usr/lib64/${soname}" \ + "/usr/lib/x86_64-linux-gnu/${soname}" \ + "/usr/lib/${soname}"; do if [ -e "${cand}" ]; then - lib="${cand}" - break + printf '%s' "${cand}" + return fi done - fi - printf '%s' "${lib}" + done } _vulkan_has_real_device() { # True if the loader enumerates a hardware GPU. vulkaninfo can exit non-zero # for unrelated reasons (no display/WSI), so key off the reported deviceType. - command -v vulkaninfo >/dev/null 2>&1 || return 0 + command -v vulkaninfo >/dev/null 2>&1 || return 1 vulkaninfo --summary 2>/dev/null | grep -qE 'PHYSICAL_DEVICE_TYPE_(DISCRETE|INTEGRATED|VIRTUAL)_GPU' } @@ -131,25 +133,24 @@ JSON echo "Real NVIDIA GPU selected; pinned Vulkan ICD to ${nvidia_lib}" return fi - echo "WARNING: ${nvidia_lib} present but no GPU enumerated; using SwiftShader." - # Surface why the NVIDIA driver did not enumerate (e.g. a missing dependency - # of libGLX_nvidia, or no render node) so the fallback is diagnosable in CI. + echo "ERROR: ${nvidia_lib} present but no GPU enumerated." + # Surface missing ICD dependencies and device-enumeration errors. + ldd "${nvidia_lib}" || true if command -v vulkaninfo >/dev/null 2>&1; then echo "--- NVIDIA Vulkan ICD diagnostic ---" VK_LOADER_DEBUG=warn vulkaninfo --summary 2>&1 | head -40 || true echo "--- end diagnostic ---" fi - unset VK_ICD_FILENAMES else - echo "WARNING: no NVIDIA Vulkan driver library found; using SwiftShader." + echo "ERROR: no NVIDIA Vulkan driver library found." fi - install_swiftshader + return 1 } VULKAN_SDK_VERSION="1.4.321.1" # The no-argument default installs SwiftShader so the existing CPU-runner CI is -# unchanged. Pass "real-gpu" to prefer a real system ICD when one is present. +# unchanged. Pass "real-gpu" to require a real system ICD. case "${1:-swiftshader}" in real-gpu) # Do not download the LunarG SDK here: its prebuilt glslc cannot run on the diff --git a/.ci/scripts/test_backend.sh b/.ci/scripts/test_backend.sh index 068d5adb260..8bfe33333a8 100755 --- a/.ci/scripts/test_backend.sh +++ b/.ci/scripts/test_backend.sh @@ -54,15 +54,8 @@ if [[ "$FLOW" == *qnn* ]]; then fi if [[ "$FLOW" == *vulkan* ]]; then - # Setup the Vulkan SDK and select an ICD: use the real system GPU ICD when one - # is present (real-GPU runner), otherwise fall back to SwiftShader (CPU - # runner). The Vulkan loader searches both standard ICD directories. - if ls /etc/vulkan/icd.d/*.json /usr/share/vulkan/icd.d/*.json \ - >/dev/null 2>&1; then - source .ci/scripts/setup-vulkan-linux-deps.sh "real-gpu" - else - source .ci/scripts/setup-vulkan-linux-deps.sh "swiftshader" - fi + # CPU runners can have Mesa ICDs installed without a usable hardware GPU. + source .ci/scripts/setup-vulkan-linux-deps.sh "swiftshader" EXTRA_BUILD_ARGS+=" -DEXECUTORCH_BUILD_VULKAN=ON" fi diff --git a/.github/workflows/vulkan.yml b/.github/workflows/vulkan.yml index 2514a8d582c..dd3de1f33fb 100644 --- a/.github/workflows/vulkan.yml +++ b/.github/workflows/vulkan.yml @@ -70,10 +70,7 @@ jobs: # not run, so setup-vulkan-linux-deps.sh sources those from conda-forge and # the system package manager instead. The NVIDIA container runtime mounts # the driver's Vulkan library but not its ICD manifest, so the script - # synthesizes one and pins the loader to it; if no NVIDIA library is found - # it falls back to SwiftShader. - # NOTE: first-run check - inspect the vulkaninfo output below to confirm a - # real NVIDIA device is selected (not llvmpipe/SwiftShader). + # synthesizes one and requires the loader to enumerate a hardware GPU. source .ci/scripts/setup-vulkan-linux-deps.sh real-gpu vulkaninfo --summary || true @@ -100,23 +97,16 @@ jobs: # Operator coverage (mirrors test-vulkan-operators-linux, on real hardware). # The custom-op prototyping binaries are GPU microbenchmarks that rely on - # GPU timestamp queries; they need a real device and crash on the - # SwiftShader software fallback. Always build them (compile coverage), but - # only run them when a real GPU was selected (setup-vulkan-linux-deps.sh - # exports ETVK_USING_SWIFTSHADER when it falls back to SwiftShader). + # GPU timestamp queries, so they require the hardware selected above. PYTHON_EXECUTABLE=python bash backends/vulkan/test/custom_ops/build_and_run.sh - if [ -z "${ETVK_USING_SWIFTSHADER:-}" ]; then - ./cmake-out/backends/vulkan/test/custom_ops/test_add - ./cmake-out/backends/vulkan/test/custom_ops/test_q8csw_linear - ./cmake-out/backends/vulkan/test/custom_ops/test_q8csw_conv2d - ./cmake-out/backends/vulkan/test/custom_ops/test_q4gsw_linear - ./cmake-out/backends/vulkan/test/custom_ops/test_choose_qparams_per_row - ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_qdq - ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_clone - ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_binary - else - echo "SwiftShader fallback active: built custom-op benchmarks but skipping execution (they require real-GPU timestamp queries)." - fi + ./cmake-out/backends/vulkan/test/custom_ops/test_add + ./cmake-out/backends/vulkan/test/custom_ops/test_q8csw_linear + ./cmake-out/backends/vulkan/test/custom_ops/test_q8csw_conv2d + ./cmake-out/backends/vulkan/test/custom_ops/test_q4gsw_linear + ./cmake-out/backends/vulkan/test/custom_ops/test_choose_qparams_per_row + ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_qdq + ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_clone + ./cmake-out/backends/vulkan/test/custom_ops/test_q8ta_binary PYTHON_EXECUTABLE=python bash backends/vulkan/test/scripts/test_op.sh --build