From 5e1cfd8e1df1d567df78999580ac2ef3f975d481 Mon Sep 17 00:00:00 2001 From: Paul Murphy Date: Tue, 15 Sep 2026 14:59:07 -0500 Subject: [PATCH 1/3] compiler-builtins: rewrite cas16 AArch64 outlined atomics with asm! The outlined atomics should follow the AAPCS64 calling conventions, and allow usage of -Z branch-protection=bti. We just need to be careful to avoid bad codegen. This generates almost identical codegen with -C opt-level=1. Those differences are entirely regalloc choices. --- .../src/aarch64_outline_atomics.rs | 59 +++++++++++++------ 1 file changed, 41 insertions(+), 18 deletions(-) diff --git a/library/compiler-builtins/compiler-builtins/src/aarch64_outline_atomics.rs b/library/compiler-builtins/compiler-builtins/src/aarch64_outline_atomics.rs index 367768ecb60d6..8c23f704aaa68 100644 --- a/library/compiler-builtins/compiler-builtins/src/aarch64_outline_atomics.rs +++ b/library/compiler-builtins/compiler-builtins/src/aarch64_outline_atomics.rs @@ -276,28 +276,51 @@ macro_rules! compare_and_swap_u128 { ($ordering:ident, $name:ident) => { intrinsics! { #[maybe_use_optimized_c_shim] - #[unsafe(naked)] pub unsafe extern "C" fn $name ( expected: u128, desired: u128, ptr: *mut u128 ) -> u128 { - core::arch::naked_asm! { - // CASP x0, x1, x2, x3, [x4]; if LSE supported. - try_lse_op!("cas", $ordering, 16, 0, 1, 2, 3, [x4]), - "mov x16, x0", - "mov x17, x1", - "0:", - // LDXP x0, x1, [x4] - concat!(ldxp!($ordering), " x0, x1, [x4]"), - "cmp x0, x16", - "ccmp x1, x17, #0, eq", - "bne 1f", - // STXP w(tmp2), x2, x3, [x4] - concat!(stxp!($ordering), " w15, x2, x3, [x4]"), - "cbnz w15, 0b", - "1:", - "ret", - have_lse = sym crate::aarch64_outline_atomics::HAVE_LSE_ATOMICS, + let mut expected_lo = (expected & 0xFFFFFFFFFFFFFFFF) as u64; + let mut expected_hi = (expected >> 64) as u64; + let desired_lo = (desired & 0xFFFFFFFFFFFFFFFF) as u64; + let desired_hi = (desired >> 64) as u64; + + unsafe { + if HAVE_LSE_ATOMICS.load(Ordering::Relaxed) != 0 { + core::arch::asm!( + ".arch_extension lse", + concat!(lse!("cas", $ordering, 16), " x0, x1, x2, x3, [x4]"), + inlateout("x0") expected_lo, + inlateout("x1") expected_hi, + in("x2") desired_lo, + in("x3") desired_hi, + in("x4") ptr, + options(nostack), + ); + } else { + core::arch::asm!( + "mov x16, x0", + "mov x17, x1", + "1:", + concat!(ldxp!($ordering), " x0, x1, [x4]"), + "cmp x0, x16", + "ccmp x1, x17, #0x0, eq", + "b.ne 2f", + concat!(stxp!($ordering), " w15, x2, x3, [x4]"), + "cbnz w15, 1b", + "2:", + inlateout("x0") expected_lo, + inlateout("x1") expected_hi, + in("x2") desired_lo, + in("x3") desired_hi, + in("x4") ptr, + out("w15") _, + out("x16") _, + out("x17") _, + options(nostack), + ); + } } + return ((expected_hi as u128) << 64) | expected_lo as u128; } } }; From 8825c49576d8eda7e484c87e1c2010f28ef96f65 Mon Sep 17 00:00:00 2001 From: Paul Murphy Date: Wed, 16 Sep 2026 11:46:46 -0500 Subject: [PATCH 2/3] compiler-builtins: rewrite cas outlined atomics with asm! The in/inlateout register operands aren't entirely accurate (w vs x). This annoys me, but it's the same register, and we cannot use a macro inside that portion of asm!. --- .../src/aarch64_outline_atomics.rs | 49 ++++++++++++------- 1 file changed, 30 insertions(+), 19 deletions(-) diff --git a/library/compiler-builtins/compiler-builtins/src/aarch64_outline_atomics.rs b/library/compiler-builtins/compiler-builtins/src/aarch64_outline_atomics.rs index 8c23f704aaa68..c4980b214e230 100644 --- a/library/compiler-builtins/compiler-builtins/src/aarch64_outline_atomics.rs +++ b/library/compiler-builtins/compiler-builtins/src/aarch64_outline_atomics.rs @@ -243,29 +243,40 @@ macro_rules! compare_and_swap { ($ordering:ident, $bytes:tt, $name:ident) => { intrinsics! { #[maybe_use_optimized_c_shim] - #[unsafe(naked)] pub unsafe extern "C" fn $name ( expected: int_ty!($bytes), desired: int_ty!($bytes), ptr: *mut int_ty!($bytes) ) -> int_ty!($bytes) { - // We can't use `AtomicI8::compare_and_swap`; we *are* compare_and_swap. - core::arch::naked_asm! { - // CAS s(0), s(1), [x2]; if LSE supported. - try_lse_op!("cas", $ordering, $bytes, 0, 1, [x2]), - // UXT s(tmp0), s(0) - concat!(uxt!($bytes), " ", reg!($bytes, 16), ", ", reg!($bytes, 0)), - "0:", - // LDXR s(0), [x2] - concat!(ldxr!($ordering, $bytes), " ", reg!($bytes, 0), ", [x2]"), - // cmp s(0), s(tmp0) - concat!("cmp ", reg!($bytes, 0), ", ", reg!($bytes, 16)), - "bne 1f", - // STXR w(tmp1), s(1), [x2] - concat!(stxr!($ordering, $bytes), " w17, ", reg!($bytes, 1), ", [x2]"), - "cbnz w17, 0b", - "1:", - "ret", - have_lse = sym crate::aarch64_outline_atomics::HAVE_LSE_ATOMICS, + let mut expected = expected; + unsafe { + if HAVE_LSE_ATOMICS.load(Ordering::Relaxed) != 0 { + core::arch::asm!( + ".arch_extension lse", + concat!(lse!("cas", $ordering, $bytes), " ", reg!($bytes, 0), ", ", reg!($bytes, 1),", [x2]"), + inlateout ( "x0" ) expected, + in("x1") desired, + in("x2") ptr, + options(nostack), + ); + } else { + core::arch::asm!( + concat!(uxt!($bytes), " ", reg!($bytes, 16), ", ", reg!($bytes, 0)), + "1:", + concat!(ldxr!($ordering, $bytes), " ", reg!($bytes, 0), ", [x2]"), + concat!("cmp ", reg!($bytes, 0), ", ", reg!($bytes, 16)), + "bne 2f", + concat!(stxr!($ordering, $bytes), " w17, ", reg!($bytes, 1), ", [x2]"), + "cbnz w17, 1b", + "2:", + inlateout("x0") expected, + in("x1") desired, + in("x2") ptr, + out("x16") _, + out("w17") _, + options(nostack), + ); + } } + expected } } }; From 982f13797d91767579100a896f5863dd909e911d Mon Sep 17 00:00:00 2001 From: Paul Murphy Date: Wed, 16 Sep 2026 15:35:25 -0500 Subject: [PATCH 3/3] compiler_builtins: rewrite remaining aarch64 outlined atomics in asm! And, remove unused macros. --- .../src/aarch64_outline_atomics.rs | 152 ++++++------------ 1 file changed, 51 insertions(+), 101 deletions(-) diff --git a/library/compiler-builtins/compiler-builtins/src/aarch64_outline_atomics.rs b/library/compiler-builtins/compiler-builtins/src/aarch64_outline_atomics.rs index c4980b214e230..161c96d9fa1ee 100644 --- a/library/compiler-builtins/compiler-builtins/src/aarch64_outline_atomics.rs +++ b/library/compiler-builtins/compiler-builtins/src/aarch64_outline_atomics.rs @@ -148,77 +148,6 @@ macro_rules! stxp { }; } -// The AArch64 assembly syntax for relocation specifiers -// when accessing symbols changes depending on the target executable format. -// In ELF (used in Linux), we have a prefix notation surrounded by colons (:specifier:sym), -// while in Mach-O object files (used in MacOS), a postfix notation is used (sym@specifier). - -/// AArch64 ELF position-independent addressing: -/// -/// adrp xN, symbol -/// add xN, xN, :lo12:symbol -/// -/// The :lo12: modifier selects the low 12 bits of the symbol address -/// and emits an ELF relocation such as R_AARCH64_ADD_ABS_LO12_NC. -/// -/// Defined by the AArch64 ELF psABI. -/// See: . -#[cfg(not(target_vendor = "apple"))] -macro_rules! sym { - ($sym:literal) => { - $sym - }; -} - -#[cfg(not(target_vendor = "apple"))] -macro_rules! sym_off { - ($sym:literal) => { - concat!(":lo12:", $sym) - }; -} - -/// Mach-O ARM64 relocation types: -/// ARM64_RELOC_PAGE21 -/// ARM64_RELOC_PAGEOFF12 -/// -/// These relocations implement the @PAGE / @PAGEOFF split used by -/// adrp + add sequences on Apple platforms. -/// -/// adrp xN, symbol@PAGE -> ARM64_RELOC_PAGE21 -/// add xN, xN, symbol@PAGEOFF -> ARM64_RELOC_PAGEOFF12 -/// -/// Relocation types defined by Apple in XNU: . -/// See: . -#[cfg(target_vendor = "apple")] -macro_rules! sym { - ($sym:literal) => { - concat!($sym, "@PAGE") - }; -} - -#[cfg(target_vendor = "apple")] -macro_rules! sym_off { - ($sym:literal) => { - concat!($sym, "@PAGEOFF") - }; -} - -// If supported, perform the requested LSE op and return, or fallthrough. -macro_rules! try_lse_op { - ($op: literal, $ordering:ident, $bytes:tt, $($reg:literal,)* [ $mem:ident ] ) => { - concat!( - ".arch_extension lse\n", - concat!("adrp x16, ", sym!("{have_lse}"), "\n"), - concat!("ldrb w16, [x16, ", sym_off!("{have_lse}"), "]\n"), - "cbz w16, 8f\n", - // LSE_OP s(reg),* [$mem] - concat!(lse!($op, $ordering, $bytes), $( " ", reg!($bytes, $reg), ", " ,)* "[", stringify!($mem), "]\n",), - "ret - 8:" - ) - }; -} - // Translate memory ordering to the LSE suffix #[rustfmt::skip] macro_rules! lse_mem_sfx { @@ -342,24 +271,35 @@ macro_rules! swap { ($ordering:ident, $bytes:tt, $name:ident) => { intrinsics! { #[maybe_use_optimized_c_shim] - #[unsafe(naked)] pub unsafe extern "C" fn $name ( left: int_ty!($bytes), right_ptr: *mut int_ty!($bytes) ) -> int_ty!($bytes) { - core::arch::naked_asm! { - // SWP s(0), s(0), [x1]; if LSE supported. - try_lse_op!("swp", $ordering, $bytes, 0, 0, [x1]), - // mov s(tmp0), s(0) - concat!("mov ", reg!($bytes, 16), ", ", reg!($bytes, 0)), - "0:", - // LDXR s(0), [x1] - concat!(ldxr!($ordering, $bytes), " ", reg!($bytes, 0), ", [x1]"), - // STXR w(tmp1), s(tmp0), [x1] - concat!(stxr!($ordering, $bytes), " w17, ", reg!($bytes, 16), ", [x1]"), - "cbnz w17, 0b", - "ret", - have_lse = sym crate::aarch64_outline_atomics::HAVE_LSE_ATOMICS, + let mut left = left; + unsafe { + if HAVE_LSE_ATOMICS.load(Ordering::Relaxed) != 0 { + core::arch::asm! { + ".arch_extension lse", + concat!( lse!("swp", $ordering, $bytes), " ", reg!($bytes, 0), ", ", reg!($bytes, 0), ", [x1]"), + inlateout("x0") left, + in("x1") right_ptr, + options(nostack), + }; + } else { + core::arch::asm! { + concat!("mov ", reg!($bytes, 16), ", ", reg!($bytes, 0)), + "1:", + concat!(ldxr!($ordering, $bytes), " ", reg!($bytes, 0), ", [x1]"), + concat!(stxr!($ordering, $bytes), " w17, ", reg!($bytes, 16), ", [x1]"), + "cbnz w17, 1b", + inlateout("x0") left, + in("x1") right_ptr, + options(nostack), + out("x16") _, + out("w17") _, + }; + } } + left } } }; @@ -370,26 +310,36 @@ macro_rules! fetch_op { ($ordering:ident, $bytes:tt, $name:ident, $op:literal, $lse_op:literal) => { intrinsics! { #[maybe_use_optimized_c_shim] - #[unsafe(naked)] pub unsafe extern "C" fn $name ( val: int_ty!($bytes), ptr: *mut int_ty!($bytes) ) -> int_ty!($bytes) { - core::arch::naked_asm! { - // LSEOP s(0), s(0), [x1]; if LSE supported. - try_lse_op!($lse_op, $ordering, $bytes, 0, 0, [x1]), - // mov s(tmp0), s(0) - concat!("mov ", reg!($bytes, 16), ", ", reg!($bytes, 0)), - "0:", - // LDXR s(0), [x1] - concat!(ldxr!($ordering, $bytes), " ", reg!($bytes, 0), ", [x1]"), - // OP s(tmp1), s(0), s(tmp0) - concat!($op, " ", reg!($bytes, 17), ", ", reg!($bytes, 0), ", ", reg!($bytes, 16)), - // STXR w(tmp2), s(tmp1), [x1] - concat!(stxr!($ordering, $bytes), " w15, ", reg!($bytes, 17), ", [x1]"), - "cbnz w15, 0b", - "ret", - have_lse = sym crate::aarch64_outline_atomics::HAVE_LSE_ATOMICS, + unsafe { + if HAVE_LSE_ATOMICS.load(Ordering::Relaxed) != 0 { + core::arch::asm! { + ".arch_extension lse", + concat!(lse!($lse_op, $ordering, $bytes), " ", reg!($bytes, 0), ", ", reg!($bytes, 0),", [x1]"), + in("x0") val, + in("x1") ptr, + options(nostack), + }; + } else { + core::arch::asm! { + concat!("mov ", reg!($bytes, 16), ", ", reg!($bytes, 0)), + "1:", + concat!(ldxr!($ordering, $bytes), " ", reg!($bytes, 0), ", [x1]"), + concat!($op, " ", reg!($bytes, 17), ", ", reg!($bytes, 0), ", ", reg!($bytes, 16)), + concat!(stxr!($ordering, $bytes), " w15, ", reg!($bytes, 17), ", [x1]"), + "cbnz w15, 1b", + in("x0") val, + in("x1") ptr, + out("w15") _, + out("x16") _, + out("x17") _, + options(nostack), + } + } } + val } } }