All of lore.kernel.org
 help / color / mirror / Atom feed
From: Vladimir Murzin <vladimir.murzin@arm.com>
To: Mark Rutland <mark.rutland@arm.com>,
	linux-arm-kernel@lists.infradead.org
Cc: ryan.roberts@arm.com, usama.anjum@arm.com, peterz@infradead.org,
	catalin.marinas@arm.com, david.laight.linux@gmail.com,
	stable@vger.kernel.org, ruanjinjie@huawei.com,
	james.morse@arm.com, yang@os.amperecomputing.com, cl@gentwo.org,
	maz@kernel.org, david@kernel.org, ljs@kernel.org,
	will@kernel.org, ardb@kernel.org
Subject: Re: [PATCH v4 20/21] arm64: percpu: Implement preemptible CMPXCHG128 ops
Date: Tue, 15 Sep 2026 15:20:38 +0100	[thread overview]
Message-ID: <625240ea-c4e2-414d-b53f-4b3d361d8b30@arm.com> (raw)
In-Reply-To: <20260908151741.394589-21-mark.rutland@arm.com>

On 9/8/26 16:17, Mark Rutland wrote:
> Use the PCPU GPR infrastructure to implement this_cpu_cmpxchg128().
> 
> Note that before this patch, the LL/SC and LSE implementation was chosen
> with an alternative branch. After this patch the implementations are
> patched inline, matching the style of the other percpu ops.
> 
> The sub-optimal register shuffling before and after this patch occurs
> because we hard-code specific registers, as LLVM (currently) lacks a way
> to place a 128-bit type in a compiler-allocated even-odd pair of
> registers for inline assembly. We should be able to improve that in
> future, given compiler support.
> 
> Test case:
> 
> | u128 outline_this_cpu_cmpxchg128(u128 __percpu *p, u128 old, u128 new)
> | {
> | 	return this_cpu_cmpxchg128(*p, old, new);
> | }
> 
> Generated code before this patch (v7.2-rc4):
> 
> | <outline_this_cpu_cmpxchg128>:
> |        paciasp
> |        stp     x29, x30, [sp, #-32]!
> |        mov     x6, x0
> |        mov     x0, x2
> |        mov     x29, sp
> |        mov     x1, x3
> |        mrs     x2, sp_el0
> |        ldr     w7, [x2, #8]
> |        add     w7, w7, #0x1
> |        str     w7, [x2, #8]
> |        mrs     x2, tpidr_el1
> |        add     x6, x6, x2
> |        b       4f
> |        mov     x2, x4
> |        mov     x3, x5
> |        mov     x4, x6
> |        mov     x5, x0
> |        mov     x7, x1
> |        casp    x0, x1, x2, x3, [x6]
> | 1:     mrs     x3, sp_el0
> |        ldr     x2, [x3, #8]
> |        sub     x2, x2, #0x1
> |        str     w2, [x3, #8]
> |        cbz     x2, 2f
> |        ldr     x2, [x3, #8]
> |        cbnz    x2, 3f
> | 2:     stp     x0, x1, [sp, #16]
> |        bl      preempt_schedule_notrace
> |        ldp     x0, x1, [sp, #16]
> | 3:     ldp     x29, x30, [sp], #32
> |        autiasp
> |        ret
> | 4:     prfm    pstl1strm, [x6]
> | 5:     ldxp    x8, x7, [x6]
> |        cmp     x8, x0
> |        ccmp    x7, x3, #0x0, eq
> |        b.ne    6f
> |        stxp    w2, x4, x5, [x6]
> |        cbnz    w2, 5b
> | 6:     mov     x0, x8
> |        mov     x1, x7
> |        b       1b
> 
> Generated code after this patch:
> 
> | <outline_this_cpu_cmpxchg128>:
> |        mov     x6, x0
> |        mov     x1, x3
> |        mov     x0, x2
> |        mov     x3, x5
> |        mrs     x7, sp_el0
> |        mov     x2, x4
> |        mov     w5, #0x10a6
> |        strh    w5, [x7, #20]
> |        mrs     x5, tpidr_el1
> |        add     x4, x6, x5
> |        prfm    pstl1strm, [x4]
> | 1:     ldxp    x9, x10, [x4]
> |        cmp     x9, x0
> |        ccmp    x10, x1, #0x0, eq
> |        b.ne    2f
> |        stxp    w8, x2, x3, [x4]
> |        cbnz    w8, 1b
> | 2:     strh    wzr, [x7, #20]
> |        mov     x0, x9
> |        mov     x1, x10
> |        ret
> 
> Signed-off-by: Mark Rutland <mark.rutland@arm.com>
> Tested-by: Muhammad Usama Anjum <usama.anjum@arm.com>
> Cc: Ada Couprie Diaz <ada.coupriediaz@arm.com>
> Cc: Ard Biesheuvel <ardb@kernel.org>
> Cc: Catalin Marinas <catalin.marinas@arm.com>
> Cc: James Morse <james.morse@arm.com>
> Cc: Jinjie Ruan <ruanjinjie@huawei.com>
> Cc: Marc Zyngier <maz@kernel.org>
> Cc: Peter Zijlstra <peterz@infradead.org>
> Cc: Vladimir Murzin <vladimir.murzin@arm.com>
> Cc: Will Deacon <will@kernel.org>
> Cc: Yang Shi <yang@os.amperecomputing.com>
> ---
>  arch/arm64/include/asm/percpu.h | 70 +++++++++++++++++++++++++++------
>  1 file changed, 57 insertions(+), 13 deletions(-)
> 
> diff --git a/arch/arm64/include/asm/percpu.h b/arch/arm64/include/asm/percpu.h
> index f19a82437b3fd..5cd98e20e825d 100644
> --- a/arch/arm64/include/asm/percpu.h
> +++ b/arch/arm64/include/asm/percpu.h
> @@ -490,19 +490,63 @@ PERCPU_CMPXCHG_OP(x,  , 64)
>  
>  #define this_cpu_cmpxchg64(pcp, o, n)	this_cpu_cmpxchg_8(pcp, o, n)
>  
> -#define this_cpu_cmpxchg128(pcp, o, n)					\
> -({									\
> -	typedef typeof(pcp) pcp_op_T__;					\
> -	u128 old__, new__, ret__;					\
> -	pcp_op_T__ *ptr__;						\
> -	old__ = o;							\
> -	new__ = n;							\
> -	preempt_disable_notrace();					\
> -	ptr__ = raw_cpu_ptr(&(pcp));					\
> -	ret__ = cmpxchg128_local((void *)ptr__, old__, new__);		\
> -	preempt_enable_notrace();					\
> -	ret__;								\
> -})
> +static inline u128
> +__percpu_cmpxchg_128(void __percpu *pcp, u128 old, u128 new)
> +{
> +	u16 *gprs = &current_thread_info()->pcpu_gprs;
> +	unsigned long addr;
> +	unsigned long off;
> +	union __u128_halves r, o = { .full = (old) },
> +			       n = { .full = (new) };
> +	register unsigned long ol asm ("x0") = o.low;
> +	register unsigned long oh asm ("x1") = o.high;
> +	register unsigned long nl asm ("x2") = n.low;
> +	register unsigned long nh asm ("x3") = n.high;
> +	unsigned long rl, rh;
> +	unsigned long tmp;
> +
> +	asm volatile (
> +	__PCPU_GPRS_BEGIN("%[gprs]", "%[pcp]", "%[off]", "%[addr]")
> +	ARM64_LSE_ATOMIC_INSN(
> +	/* LL/SC */
> +       "       prfm    pstl1strm, [%[addr]]\n"
> +       "1:     ldxp    %[rl], %[rh], [%[addr]]\n"
> +       "       cmp     %[rl], %[ol]\n"
> +       "       ccmp    %[rh], %[oh], 0, eq\n"
> +       "       b.ne    2f\n"
> +       "       stxp    %w[tmp], %[nl], %[nh], [%[addr]]\n"
> +       "       cbnz    %w[tmp], 1b\n"
> +       "2:\n"
> +	,
> +	/* LSE atomics */
> +	"	casp	%[ol], %[oh], %[nl], %[nh], [%[addr]]\n"
> +	"	mov	%[rl], %[ol]\n"
> +	"	mov	%[rh], %[oh]\n"
> +	__nops(4)
> +	)
> +	__PCPU_GPRS_END("%[gprs]")
> +	: [gprs] "=Qo" (*gprs),
> +	  [addr] "=&r" (addr),
> +	  [off] "=&r" (off),
> +	  [tmp] "=&r" (tmp),
> +	  [ol] "+&r" (ol),
> +	  [oh] "+&r" (oh),
> +	  [rl] "=&r" (rl),
> +	  [rh] "=&r" (rh)
> +	: [pcp] "r" (pcp),
> +	  [nl] "r" (nl),
> +	  [nh] "r" (nh)
> +	: "memory", "cc"
> +	);
> +
> +	r.low = rl;
> +	r.high = rh;
> +
> +	return r.full;
> +}
> +
> +#define this_cpu_cmpxchg128(pcp, o, n)	\
> +	_pcp_wrap_return(__percpu_cmpxchg_128, pcp, o, n)
>  
>  #ifdef __KVM_NVHE_HYPERVISOR__
>  extern unsigned long __hyp_per_cpu_offset(unsigned int cpu);
> -- 2.30.2
> 

FWIW,

Reviewed-by: Vladimir Murzin <vladimir.murzin@arm.com>



  reply	other threads:[~2026-09-15 14:20 UTC|newest]

Thread overview: 44+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-08 15:17 [PATCH v4 00/21] Preemptible this_cpu_*() operations Mark Rutland
2026-09-08 15:17 ` [PATCH v4 01/21] arm64: percpu: Fix this_cpu_write() casting Mark Rutland
2026-09-10 10:56   ` Lorenzo Stoakes (ARM)
2026-09-08 15:17 ` [PATCH v4 02/21] arm64: percpu: Fix this_cpu_and() mask generation Mark Rutland
2026-09-08 15:17 ` [PATCH v4 03/21] arm64: percpu: Fix LSE operations on {8,16}-bit types Mark Rutland
2026-09-10 15:18   ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 04/21] arm64: cmpxchg: LL/SC: Avoid redundant extension Mark Rutland
2026-09-15 14:21   ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 05/21] arm64: cmpxchg128: LSE: Remove redundant operands Mark Rutland
2026-09-10 10:06   ` Vladimir Murzin
2026-09-11 11:43     ` Mark Rutland
2026-09-08 15:17 ` [PATCH v4 06/21] arm64: preempt: Simplify and optimize __preempt_count_dec_and_test() Mark Rutland
2026-09-08 15:17 ` [PATCH v4 07/21] arm64: preempt: Treat should_resched() as unlikely Mark Rutland
2026-09-08 15:17 ` [PATCH v4 08/21] arm64: ptrace: Always inline pt_regs_[read,write}_reg() Mark Rutland
2026-09-11 10:20   ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 09/21] arm64: percpu: Factor out percpu offset asm Mark Rutland
2026-09-11 12:39   ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 10/21] arm64: gpr-num: Add wxN aliases for wN registers Mark Rutland
2026-09-11 12:43   ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 11/21] arm64: gpr-num: add __GPR_NUM() helper Mark Rutland
2026-09-10 14:24   ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 12/21] arm64: entry: sdei: Restore all clobberable GPRs Mark Rutland
2026-09-11 12:47   ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 13/21] arm64: entry: sdei: Make 'tsk' available Mark Rutland
2026-09-10 13:06   ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 14/21] arm64: percpu: Add infrastructure for preemptible this_cpu_*() ops Mark Rutland
2026-09-15 14:19   ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 15/21] arm64: percpu: Implement preemptible read/write ops Mark Rutland
2026-09-15 11:58   ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 16/21] arm64: percpu: Implement preemptible void RMW ops Mark Rutland
2026-09-15 12:01   ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 17/21] arm64: percpu: Implement preemptible return " Mark Rutland
2026-09-15 12:03   ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 18/21] arm64: percpu: Implement preemptible XCHG ops Mark Rutland
2026-09-15 12:07   ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 19/21] arm64: percpu: Implement preemptible CMPXCHG ops Mark Rutland
2026-09-15 14:20   ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 20/21] arm64: percpu: Implement preemptible CMPXCHG128 ops Mark Rutland
2026-09-15 14:20   ` Vladimir Murzin [this message]
2026-09-08 15:17 ` [PATCH v4 21/21] arm64: percpu: Remove _pcp_protect*() wrappers Mark Rutland
2026-09-15 14:21   ` Vladimir Murzin
2026-09-11 13:44 ` [PATCH v4 00/21] Preemptible this_cpu_*() operations Will Deacon
2026-09-24 16:03 ` (subset) " Catalin Marinas
2026-09-30 13:14   ` Muhammad Usama Anjum

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=625240ea-c4e2-414d-b53f-4b3d361d8b30@arm.com \
    --to=vladimir.murzin@arm.com \
    --cc=ardb@kernel.org \
    --cc=catalin.marinas@arm.com \
    --cc=cl@gentwo.org \
    --cc=david.laight.linux@gmail.com \
    --cc=david@kernel.org \
    --cc=james.morse@arm.com \
    --cc=linux-arm-kernel@lists.infradead.org \
    --cc=ljs@kernel.org \
    --cc=mark.rutland@arm.com \
    --cc=maz@kernel.org \
    --cc=peterz@infradead.org \
    --cc=ruanjinjie@huawei.com \
    --cc=ryan.roberts@arm.com \
    --cc=stable@vger.kernel.org \
    --cc=usama.anjum@arm.com \
    --cc=will@kernel.org \
    --cc=yang@os.amperecomputing.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.