diff --git a/_typos.toml b/_typos.toml deleted file mode 100644 index c4918b11462..00000000000 --- a/_typos.toml +++ /dev/null @@ -1,9 +0,0 @@ -[default.extend-words] -ba = "ba" -hsa = "hsa" -olt = "olt" -seh = "seh" -typ = "typ" - -[files] -extend-exclude = ["src/intrinsic/archs.rs", "src/intrinsic/old_archs.rs"] diff --git a/src/asm.rs b/src/asm.rs index 817acb98218..3e571c9f9ef 100644 --- a/src/asm.rs +++ b/src/asm.rs @@ -1,5 +1,3 @@ -// cSpell:ignoreRegExp [afkspqvwy]reg - use std::borrow::Cow; use std::fmt::Write; diff --git a/src/attributes.rs b/src/attributes.rs index 2e9bf428741..b6b0352d251 100644 --- a/src/attributes.rs +++ b/src/attributes.rs @@ -18,6 +18,8 @@ use rustc_target::spec::Arch; #[cfg(feature = "master")] use crate::base; use crate::context::CodegenCx; +#[cfg(feature = "master")] +use crate::gcc_util::to_gcc_aarch64_extension; use crate::gcc_util::to_gcc_features; /// Checks if the function `instance` is recursively inline. @@ -146,6 +148,21 @@ pub fn from_fn_attrs<'gcc, 'tcx>( } } + #[cfg(feature = "master")] + if cx.sess().target.arch == Arch::AArch64 { + let extensions = codegen_fn_attrs + .target_features + .iter() + .filter_map(|feature| to_gcc_aarch64_extension(feature.name.as_str())) + .map(|extension| format!("+{extension}")) + .collect::>() + .join(","); + if !extensions.is_empty() { + func.add_attribute(FnAttribute::Target(&extensions)); + } + return; + } + #[cfg(feature = "master")] let x86_interrupt = is_x86_interrupt(fn_abi); #[cfg(not(feature = "master"))] @@ -197,7 +214,8 @@ pub fn from_fn_attrs<'gcc, 'tcx>( Arch::X86 | Arch::X86_64 | Arch::PowerPC => { func.add_attribute(FnAttribute::Target(&target_features)) } - // The target attribute is not supported on other targets in GCC. + // FIXME: GCC also supports the target attribute on other targets, with their own + // spellings. _ => (), } } diff --git a/src/back/lto.rs b/src/back/lto.rs index baf1fda02e2..6ccfe6f0e6b 100644 --- a/src/back/lto.rs +++ b/src/back/lto.rs @@ -11,12 +11,10 @@ // does not remove it? // // FIXME(antoyo): for performance, check which optimizations the C++ frontend enables. -// cSpell:disable // Fix these warnings: // /usr/bin/ld: warning: type of symbol `_RNvNvNvNtCs5JWOrf9uCus_5rayon11thread_pool19WORKER_THREAD_STATE7___getit5___KEY' changed from 1 to 6 in /tmp/ccKeUSiR.ltrans0.ltrans.o // /usr/bin/ld: warning: type of symbol `_RNvNvNvNvNtNtNtCsAj5i4SGTR7_3std4sync4mpmc5waker17current_thread_id5DUMMY7___getit5___KEY' changed from 1 to 6 in /tmp/ccKeUSiR.ltrans0.ltrans.o // /usr/bin/ld: warning: incremental linking of LTO and non-LTO objects; using -flinker-output=nolto-rel which will bypass whole program optimization -// cSpell:enable use std::ffi::CString; use std::fs::{self, File}; use std::path::{Path, PathBuf}; diff --git a/src/back/write.rs b/src/back/write.rs index 1f4fd8a314a..dcb4bcde38b 100644 --- a/src/back/write.rs +++ b/src/back/write.rs @@ -131,7 +131,6 @@ pub(crate) fn codegen( if fat_lto { let lto_path = format!("{}.lto", path); - // cSpell:disable // FIXME(antoyo): The LTO frontend generates the following warning: // ../build_sysroot/sysroot_src/library/core/src/num/dec2flt/lemire.rs:150:15: warning: type of ā€˜_ZN4core3num7dec2flt5table17POWER_OF_FIVE_12817ha449a68fb31379e4E’ does not match original declaration [-Wlto-type-mismatch] // 150 | let (lo5, hi5) = POWER_OF_FIVE_128[index]; @@ -139,7 +138,6 @@ pub(crate) fn codegen( // lto1: note: ā€˜_ZN4core3num7dec2flt5table17POWER_OF_FIVE_12817ha449a68fb31379e4E’ was previously declared here // // This option is to mute it to make the UI tests pass with LTO enabled. - // cSpell:enable context.add_driver_option("-Wno-lto-type-mismatch"); // NOTE: this doesn't actually generate an executable. With the above // flags, it combines the .o files together in another .o. diff --git a/src/gcc_util.rs b/src/gcc_util.rs index 34544751a4d..5ecc971830f 100644 --- a/src/gcc_util.rs +++ b/src/gcc_util.rs @@ -67,7 +67,6 @@ pub(crate) fn global_gcc_features(sess: &EarlySession) -> Vec { // To find a list of GCC's names, check https://gcc.gnu.org/onlinedocs/gcc/Function-Attributes.html pub fn to_gcc_features<'a>(target: &Target, s: &'a str) -> SmallVec<[&'a str; 2]> { - // cSpell:disable match (&target.arch, s) { // FIXME: seems like x87 does not exist? (&Arch::X86 | &Arch::X86_64, "x87") => smallvec![], @@ -111,7 +110,77 @@ pub fn to_gcc_features<'a>(target: &Target, s: &'a str) -> SmallVec<[&'a str; 2] (&Arch::AArch64, "sve2-bitperm") => smallvec!["sve2-bitperm", "neon"], (_, s) => smallvec![s], } - // cSpell:enable +} + +/// Translate a Rust AArch64 feature name to the GCC extension enabled by `target("+name")`. +/// GCC rejects unknown extensions, so features without one return `None`. +#[cfg(feature = "master")] +pub fn to_gcc_aarch64_extension(feature: &str) -> Option<&'static str> { + let extension = match feature { + "neon" => "simd", + "rdm" => "rdma", + "fhm" => "fp16fml", + "jsconv" => "jscvt", + "mte" => "memtag", + "rand" => "rng", + "spe" => "profile", + "paca" | "pacg" => "pauth", + "aes" => "aes", + "bf16" => "bf16", + "crc" => "crc", + "cssc" => "cssc", + "dotprod" => "dotprod", + "f32mm" => "f32mm", + "f64mm" => "f64mm", + "faminmax" => "faminmax", + "fcma" => "fcma", + "flagm" => "flagm", + "flagm2" => "flagm2", + "fp8" => "fp8", + "fp8dot2" => "fp8dot2", + "fp8dot4" => "fp8dot4", + "fp8fma" => "fp8fma", + "fp16" => "fp16", + "frintts" => "frintts", + "i8mm" => "i8mm", + "lse" => "lse", + "lse128" => "lse128", + "lut" => "lut", + "mops" => "mops", + "rcpc" => "rcpc", + "rcpc2" => "rcpc2", + "rcpc3" => "rcpc3", + "sb" => "sb", + "sha2" => "sha2", + "sha3" => "sha3", + "sm4" => "sm4", + "sme" => "sme", + "sme-b16b16" => "sme-b16b16", + "sme-f8f16" => "sme-f8f16", + "sme-f8f32" => "sme-f8f32", + "sme-f16f16" => "sme-f16f16", + "sme-f64f64" => "sme-f64f64", + "sme-i16i64" => "sme-i16i64", + "sme-lutv2" => "sme-lutv2", + "sme2" => "sme2", + "sme2p1" => "sme2p1", + "ssbs" => "ssbs", + "ssve-fp8dot2" => "ssve-fp8dot2", + "ssve-fp8dot4" => "ssve-fp8dot4", + "ssve-fp8fma" => "ssve-fp8fma", + "sve" => "sve", + "sve-b16b16" => "sve-b16b16", + "sve2" => "sve2", + "sve2-aes" => "sve2-aes", + "sve2-bitperm" => "sve2-bitperm", + "sve2-sha3" => "sve2-sha3", + "sve2-sm4" => "sve2-sm4", + "sve2p1" => "sve2p1", + "tme" => "tme", + "wfxt" => "wfxt", + _ => return None, + }; + Some(extension) } /// Translate a Rust feature name to the name libgccjit reports in its target diff --git a/src/int.rs b/src/int.rs index de8762d1c06..cc82bf01f81 100644 --- a/src/int.rs +++ b/src/int.rs @@ -2,8 +2,6 @@ //! This module exists because some integer types are not supported on some gcc platforms, e.g. //! 128-bit integers on 32-bit platforms and thus require to be handled manually. -// cSpell:words cmpti divti modti mulodi muloti udivti umodti - use gccjit::{ BinaryOp, CType, ComparisonOp, FunctionType, Location, RValue, ToRValue, Type, UnaryOp, }; @@ -958,11 +956,9 @@ impl<'gcc, 'tcx> CodegenCx<'gcc, 'tcx> { let value = self.int_to_float_cast(signed, value, self.float_type); return self.context.new_cast(None, value, dest_typ); } - // cSpell:disable TypeKind::Float => "tisf", TypeKind::Double => "tidf", TypeKind::FP128 => "titf", - // cSpell:enable kind => panic!("cannot cast a non-native integer to type {:?}", kind), }; let sign = if signed { "" } else { "un" }; @@ -1008,12 +1004,10 @@ impl<'gcc, 'tcx> CodegenCx<'gcc, 'tcx> { _ => (None, value_type), }; let name_suffix = match self.type_kind(value_type) { - // cSpell:disable // Since we will cast Half to a float, we use sfti for both. TypeKind::Half | TypeKind::Float => "sfti", TypeKind::Double => "dfti", TypeKind::FP128 => "tfti", - // cSpell:enable kind => panic!("cannot cast a {:?} to non-native integer", kind), }; let sign = if signed { "" } else { "uns" }; diff --git a/src/intrinsic/llvm.rs b/src/intrinsic/llvm.rs index d32fd10bdb9..100c1792c05 100644 --- a/src/intrinsic/llvm.rs +++ b/src/intrinsic/llvm.rs @@ -1716,11 +1716,30 @@ pub fn intrinsic<'gcc, 'tcx>(name: &str, cx: &CodegenCx<'gcc, 'tcx>) -> Function "llvm.aarch64.neon.addp.v2i32" => "__builtin_aarch64_addpv2si", "llvm.aarch64.neon.fcvtau.v4i32.v4f32" => "__builtin_aarch64_lrounduv4sfv4si_us", "llvm.aarch64.neon.fmin.v4f32" => "__builtin_aarch64_fmin_nanv4sf", + "llvm.aarch64.neon.sqrdmulh.v4i16" => "__builtin_aarch64_sqrdmulhv4hi", + "llvm.aarch64.neon.sqrdmulh.v8i16" => "__builtin_aarch64_sqrdmulhv8hi", + "llvm.aarch64.neon.sqrdmulh.v2i32" => "__builtin_aarch64_sqrdmulhv2si", "llvm.aarch64.neon.sqrdmulh.v4i32" => "__builtin_aarch64_sqrdmulhv4si", + "llvm.aarch64.neon.sqrdmlah.v4i16" => "__builtin_aarch64_sqrdmlahv4hi", + "llvm.aarch64.neon.sqrdmlah.v8i16" => "__builtin_aarch64_sqrdmlahv8hi", + "llvm.aarch64.neon.sqrdmlah.v2i32" => "__builtin_aarch64_sqrdmlahv2si", + "llvm.aarch64.neon.sqrdmlah.v4i32" => "__builtin_aarch64_sqrdmlahv4si", + "llvm.aarch64.neon.sqrdmlsh.v4i16" => "__builtin_aarch64_sqrdmlshv4hi", + "llvm.aarch64.neon.sqrdmlsh.v8i16" => "__builtin_aarch64_sqrdmlshv8hi", + "llvm.aarch64.neon.sqrdmlsh.v2i32" => "__builtin_aarch64_sqrdmlshv2si", + "llvm.aarch64.neon.sqrdmlsh.v4i32" => "__builtin_aarch64_sqrdmlshv4si", "llvm.aarch64.neon.sqxtn.v4i16" => "__builtin_aarch64_sqmovnv4si", "llvm.aarch64.neon.sqxtun.v4i16" => "__builtin_aarch64_sqmovunv4si_us", "llvm.aarch64.neon.uqxtn.v8i8" => "__builtin_aarch64_uqmovnv8hi", "llvm.aarch64.neon.sqshrun.v4i16" => "__builtin_aarch64_sqshrun_nv4si", + "llvm.aarch64.crc32b" => "__builtin_aarch64_crc32b", + "llvm.aarch64.crc32h" => "__builtin_aarch64_crc32h", + "llvm.aarch64.crc32w" => "__builtin_aarch64_crc32w", + "llvm.aarch64.crc32x" => "__builtin_aarch64_crc32x", + "llvm.aarch64.crc32cb" => "__builtin_aarch64_crc32cb", + "llvm.aarch64.crc32ch" => "__builtin_aarch64_crc32ch", + "llvm.aarch64.crc32cw" => "__builtin_aarch64_crc32cw", + "llvm.aarch64.crc32cx" => "__builtin_aarch64_crc32cx", // NOTE: this file is generated by https://github.com/GuillaumeGomez/llvmint/blob/master/generate_list.py _ => map_arch_intrinsic(name), diff --git a/src/lib.rs b/src/lib.rs index a006ca26e24..9e9902579e7 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -1,12 +1,10 @@ /* * FIXME(antoyo): implement equality in libgccjit based on https://zpz.github.io/blog/overloading-equality-operator-in-cpp-class-hierarchy/ (for type equality?) * For Thin LTO, this might be helpful: -// cspell:disable-next-line * In gcc 4.6 -fwhopr was removed and became default with -flto. The non-whopr path can still be executed via -flto-partition=none. * Or the new incremental LTO (https://www.phoronix.com/news/GCC-Incremental-LTO-Patches)? * * Maybe some missing optimizations enabled by rustc's LTO is in there: https://gcc.gnu.org/onlinedocs/gcc/Optimize-Options.html -// cspell:disable-next-line * Like -fipa-icf (should be already enabled) and maybe -fdevirtualize-at-ltrans. * FIXME: disable debug info always being emitted. Perhaps this slows down things? * @@ -478,14 +476,12 @@ fn target_config(sess: &EarlySession, target_info: &SharedTargetInfo) -> TargetC |feature| { let gccjit_feature_name = to_gcc_target_info_feature(sess, feature); target_info.cpu_supports(gccjit_feature_name) - // cSpell:disable /* adx, aes, avx, avx2, avx512bf16, avx512bitalg, avx512bw, avx512cd, avx512dq, avx512er, avx512f, avx512fp16, avx512ifma, avx512pf, avx512vbmi, avx512vbmi2, avx512vl, avx512vnni, avx512vp2intersect, avx512vpopcntdq, bmi1, bmi2, cmpxchg16b, ermsb, f16c, fma, fxsr, gfni, lzcnt, movbe, pclmulqdq, popcnt, rdrand, rdseed, rtm, sha, sse, sse2, sse3, sse4.1, sse4.2, sse4a, ssse3, tbm, vaes, vpclmulqdq, xsave, xsavec, xsaveopt, xsaves */ - // cSpell:enable }, ); diff --git a/tests/run/aarch64_crc32.rs b/tests/run/aarch64_crc32.rs new file mode 100644 index 00000000000..0aee3059479 --- /dev/null +++ b/tests/run/aarch64_crc32.rs @@ -0,0 +1,80 @@ +// Compiler: +// +// Run-time: +// status: 0 + +// The CRC32 intrinsics only compile when `#[target_feature(enable = "crc")]` reaches GCC as +// `target("+crc")`; a wrong lowering returns a different checksum. + +#[cfg(target_arch = "aarch64")] +mod crc32 { + use std::arch::aarch64::*; + + const IEEE_POLYNOMIAL: u64 = 0xEDB8_8320; + const CASTAGNOLI_POLYNOMIAL: u64 = 0x82F6_3B78; + + fn software_crc32(crc: u32, data: u64, bits: u32, polynomial: u64) -> u32 { + let mut state = crc as u64 ^ data; + for _ in 0..bits { + state = if state & 1 == 1 { (state >> 1) ^ polynomial } else { state >> 1 }; + } + state as u32 + } + + #[target_feature(enable = "crc")] + fn check(crc: u32, data: u64) { + for (polynomial, byte, half, word, double) in [ + ( + IEEE_POLYNOMIAL, + __crc32b(crc, data as u8), + __crc32h(crc, data as u16), + __crc32w(crc, data as u32), + __crc32d(crc, data), + ), + ( + CASTAGNOLI_POLYNOMIAL, + __crc32cb(crc, data as u8), + __crc32ch(crc, data as u16), + __crc32cw(crc, data as u32), + __crc32cd(crc, data), + ), + ] { + assert_eq!(byte, software_crc32(crc, data as u8 as u64, 8, polynomial)); + assert_eq!(half, software_crc32(crc, data as u16 as u64, 16, polynomial)); + assert_eq!(word, software_crc32(crc, data as u32 as u64, 32, polynomial)); + assert_eq!(double, software_crc32(crc, data, 64, polynomial)); + } + } + + #[target_feature(enable = "crc")] + fn check_values() { + let input = b"123456789"; + let ieee = !input.iter().fold(!0, |crc, &byte| __crc32b(crc, byte)); + let castagnoli = !input.iter().fold(!0, |crc, &byte| __crc32cb(crc, byte)); + assert_eq!(ieee, 0xCBF4_3926); + assert_eq!(castagnoli, 0xE306_9283); + } + + pub fn run() { + assert!( + std::arch::is_aarch64_feature_detected!("crc"), + "this test needs a CPU with the `crc` feature" + ); + let inputs = [ + (0, 0), + (!0, !0), + (0x1234_5678, 0x0123_4567_89AB_CDEF), + (0xDEAD_BEEF, 0xFEDC_BA98_7654_3210), + (0x8000_0001, 0x8000_0000_0000_0001), + ]; + for (crc, data) in inputs { + unsafe { check(crc, data) }; + } + unsafe { check_values() }; + } +} + +fn main() { + #[cfg(target_arch = "aarch64")] + crc32::run(); +} diff --git a/tests/run/aarch64_rdma.rs b/tests/run/aarch64_rdma.rs new file mode 100644 index 00000000000..cb1bf441842 --- /dev/null +++ b/tests/run/aarch64_rdma.rs @@ -0,0 +1,106 @@ +// Compiler: +// +// Run-time: +// status: 0 + +// The saturating rounding doubling multiply intrinsics must match the ARM pseudocode in every +// lane, including the saturating corner cases; the accumulating forms need `+rdma` to compile. + +#[cfg(target_arch = "aarch64")] +mod rdma { + use std::arch::aarch64::*; + use std::mem::transmute; + + const VALUES_16: [i16; 8] = [i16::MIN, i16::MIN + 1, -12345, -1, 0, 1, 23456, i16::MAX]; + const VALUES_32: [i32; 8] = + [i32::MIN, i32::MIN + 1, -123_456_789, -1, 0, 1, 987_654_321, i32::MAX]; + + fn reference(accumulator: i64, left: i64, right: i64, bits: u32, sign: i128) -> i64 { + let rounding = 1i128 << (bits - 1); + let value = + ((accumulator as i128) << bits) + sign * 2 * left as i128 * right as i128 + rounding; + let maximum = (1i128 << (bits - 1)) - 1; + (value >> bits).clamp(-maximum - 1, maximum) as i64 + } + + macro_rules! check { + ($name:ident, $element:ty, $lanes:literal, $multiply:ident, $add:ident, $subtract:ident) => { + #[target_feature(enable = "rdm")] + fn $name(values: &[$element]) { + let count = values.len(); + let bits = <$element>::BITS; + for first in 0..count { + for second in 0..count { + for third in 0..count { + let accumulator: [$element; $lanes] = + std::array::from_fn(|lane| values[(first + lane) % count]); + let left: [$element; $lanes] = + std::array::from_fn(|lane| values[(second + 2 * lane) % count]); + let right: [$element; $lanes] = + std::array::from_fn(|lane| values[(third + 3 * lane) % count]); + let (multiplied, added, subtracted): ( + [$element; $lanes], + [$element; $lanes], + [$element; $lanes], + ) = unsafe { + ( + transmute($multiply(transmute(left), transmute(right))), + transmute($add( + transmute(accumulator), + transmute(left), + transmute(right), + )), + transmute($subtract( + transmute(accumulator), + transmute(left), + transmute(right), + )), + ) + }; + for lane in 0..$lanes { + let accumulator = accumulator[lane] as i64; + let left = left[lane] as i64; + let right = right[lane] as i64; + assert_eq!( + multiplied[lane] as i64, + reference(0, left, right, bits, 1) + ); + assert_eq!( + added[lane] as i64, + reference(accumulator, left, right, bits, 1) + ); + assert_eq!( + subtracted[lane] as i64, + reference(accumulator, left, right, bits, -1) + ); + } + } + } + } + } + }; + } + + check!(check_int16x4, i16, 4, vqrdmulh_s16, vqrdmlah_s16, vqrdmlsh_s16); + check!(check_int16x8, i16, 8, vqrdmulhq_s16, vqrdmlahq_s16, vqrdmlshq_s16); + check!(check_int32x2, i32, 2, vqrdmulh_s32, vqrdmlah_s32, vqrdmlsh_s32); + check!(check_int32x4, i32, 4, vqrdmulhq_s32, vqrdmlahq_s32, vqrdmlshq_s32); + + pub fn run() { + assert!( + std::arch::is_aarch64_feature_detected!("rdm"), + "this test needs a CPU with the `rdm` feature" + ); + unsafe { + check_int16x4(&VALUES_16); + check_int16x8(&VALUES_16); + check_int32x2(&VALUES_32); + check_int32x4(&VALUES_32); + } + } +} + +fn main() { + #[cfg(target_arch = "aarch64")] + rdma::run(); +}