From mboxrd@z Thu Jan 1 00:00:00 1970 Return-Path: X-Spam-Checker-Version: SpamAssassin 3.4.0 (2014-02-07) on aws-us-west-2-korg-lkml-1.web.codeaurora.org Received: from bombadil.infradead.org (bombadil.infradead.org [198.137.202.133]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.lore.kernel.org (Postfix) with ESMTPS id 18BD7C79F82 for ; Tue, 8 Sep 2026 15:25:37 +0000 (UTC) DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; d=lists.infradead.org; s=bombadil.20210309; h=Sender:Cc:List-Subscribe: List-Help:List-Post:List-Archive:List-Unsubscribe:List-Id: Content-Transfer-Encoding:MIME-Version:References:In-Reply-To:Message-Id:Date :Subject:To:From:Reply-To:Content-Type:Content-ID:Content-Description: Resent-Date:Resent-From:Resent-Sender:Resent-To:Resent-Cc:Resent-Message-ID: List-Owner; bh=M4P7vyCOVni+ZsK8pF/yuS7E8KI5hqR5gtPw7mD705Y=; b=pp17TaA84hH4mV q5QJPC94T7Q6mg2Dec7A3UlW/lBFVcAbh0LmnyLvFLfmwlqcM5L+6wgPXVNTyW0xaDwoj0Y2X3pFh xz3Qm+FDhyVUSi+4MQVc1qH7Iz5J6Fbj9bx49CxhhaTtP9C/gKEGevgXtWBkgUtxMU96mzpeu282P 5bAZfkAg8O31YF0qQMz1nBq92evAlmtfZJDt7gIBW9C/kKjosw8aRLybtuC5ej9GMEn3OmgwWQIVG ZoyQGBrAxnMLA1+wJgJnvcgiudmIO6ZV6fGwn6b+4Z9CM/tjy6eRMs60A14i4Q+fVXXURKfoPiEnu zYjxmfCcXPNCq6wZ50Tw==; Received: from localhost ([::1] helo=bombadil.infradead.org) by bombadil.infradead.org with esmtp (Exim 4.99.1 #2 (Red Hat Linux)) id 1x3xhS-00000009TcD-0fhG; Tue, 08 Sep 2026 15:25:30 +0000 Received: from foss.arm.com ([217.140.110.172]) by bombadil.infradead.org with esmtp (Exim 4.99.1 #2 (Red Hat Linux)) id 1x3xhP-00000009Tay-0Hl7 for linux-arm-kernel@lists.infradead.org; Tue, 08 Sep 2026 15:25:28 +0000 Received: from usa-sjc-imap-foss1.foss.arm.com (unknown [10.121.207.14]) by usa-sjc-mx-foss1.foss.arm.com (Postfix) with ESMTP id AE54E1476; Tue, 8 Sep 2026 08:25:22 -0700 (PDT) Received: from lakrids.cambridge.arm.com (usa-sjc-imap-foss1.foss.arm.com [10.121.207.14]) by usa-sjc-imap-foss1.foss.arm.com (Postfix) with ESMTPA id 98C453F7B4; Tue, 8 Sep 2026 08:25:23 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=simple/simple; d=arm.com; s=foss; t=1788881126; bh=ulS9e57JOpVuxlf/+qzYMC41G5m8IkihwUPBziATUvc=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=Lu4bxN2C5Mdu1SbfESGO2/QbLqhMv7PYpc1LFkXjw0uVrCkHm6mRMS2NvZNmQQFWs NtBVvqaBnMvj4qMpgTS3wtl54yz50U9EVnw+pJpSpT21Frrm6dafb5kZZStTDxV7dU zS5d31sX5OUGa0hT/skPpP84sl5o1jUjwQ515E4k= From: Mark Rutland To: linux-arm-kernel@lists.infradead.org Subject: [PATCH v4 20/21] arm64: percpu: Implement preemptible CMPXCHG128 ops Date: Tue, 8 Sep 2026 16:17:40 +0100 Message-Id: <20260908151741.394589-21-mark.rutland@arm.com> X-Mailer: git-send-email 2.30.2 In-Reply-To: <20260908151741.394589-1-mark.rutland@arm.com> References: <20260908151741.394589-1-mark.rutland@arm.com> MIME-Version: 1.0 Content-Transfer-Encoding: 8bit X-CRM114-Version: 20100106-BlameMichelson ( TRE 0.9.0 (BSD) ) MR-646709E3 X-CRM114-CacheID: sfid-20260908_082527_176645_7588F244 X-CRM114-Status: GOOD ( 12.64 ) X-BeenThere: linux-arm-kernel@lists.infradead.org X-Mailman-Version: 2.1.34 Precedence: list List-Id: List-Unsubscribe: , List-Archive: List-Post: List-Help: List-Subscribe: , Cc: mark.rutland@arm.com, vladimir.murzin@arm.com, ryan.roberts@arm.com, usama.anjum@arm.com, peterz@infradead.org, catalin.marinas@arm.com, david.laight.linux@gmail.com, stable@vger.kernel.org, ruanjinjie@huawei.com, james.morse@arm.com, yang@os.amperecomputing.com, cl@gentwo.org, maz@kernel.org, david@kernel.org, ljs@kernel.org, will@kernel.org, ardb@kernel.org Sender: "linux-arm-kernel" Errors-To: linux-arm-kernel-bounces+linux-arm-kernel=archiver.kernel.org@lists.infradead.org Use the PCPU GPR infrastructure to implement this_cpu_cmpxchg128(). Note that before this patch, the LL/SC and LSE implementation was chosen with an alternative branch. After this patch the implementations are patched inline, matching the style of the other percpu ops. The sub-optimal register shuffling before and after this patch occurs because we hard-code specific registers, as LLVM (currently) lacks a way to place a 128-bit type in a compiler-allocated even-odd pair of registers for inline assembly. We should be able to improve that in future, given compiler support. Test case: | u128 outline_this_cpu_cmpxchg128(u128 __percpu *p, u128 old, u128 new) | { | return this_cpu_cmpxchg128(*p, old, new); | } Generated code before this patch (v7.2-rc4): | : | paciasp | stp x29, x30, [sp, #-32]! | mov x6, x0 | mov x0, x2 | mov x29, sp | mov x1, x3 | mrs x2, sp_el0 | ldr w7, [x2, #8] | add w7, w7, #0x1 | str w7, [x2, #8] | mrs x2, tpidr_el1 | add x6, x6, x2 | b 4f | mov x2, x4 | mov x3, x5 | mov x4, x6 | mov x5, x0 | mov x7, x1 | casp x0, x1, x2, x3, [x6] | 1: mrs x3, sp_el0 | ldr x2, [x3, #8] | sub x2, x2, #0x1 | str w2, [x3, #8] | cbz x2, 2f | ldr x2, [x3, #8] | cbnz x2, 3f | 2: stp x0, x1, [sp, #16] | bl preempt_schedule_notrace | ldp x0, x1, [sp, #16] | 3: ldp x29, x30, [sp], #32 | autiasp | ret | 4: prfm pstl1strm, [x6] | 5: ldxp x8, x7, [x6] | cmp x8, x0 | ccmp x7, x3, #0x0, eq | b.ne 6f | stxp w2, x4, x5, [x6] | cbnz w2, 5b | 6: mov x0, x8 | mov x1, x7 | b 1b Generated code after this patch: | : | mov x6, x0 | mov x1, x3 | mov x0, x2 | mov x3, x5 | mrs x7, sp_el0 | mov x2, x4 | mov w5, #0x10a6 | strh w5, [x7, #20] | mrs x5, tpidr_el1 | add x4, x6, x5 | prfm pstl1strm, [x4] | 1: ldxp x9, x10, [x4] | cmp x9, x0 | ccmp x10, x1, #0x0, eq | b.ne 2f | stxp w8, x2, x3, [x4] | cbnz w8, 1b | 2: strh wzr, [x7, #20] | mov x0, x9 | mov x1, x10 | ret Signed-off-by: Mark Rutland Tested-by: Muhammad Usama Anjum Cc: Ada Couprie Diaz Cc: Ard Biesheuvel Cc: Catalin Marinas Cc: James Morse Cc: Jinjie Ruan Cc: Marc Zyngier Cc: Peter Zijlstra Cc: Vladimir Murzin Cc: Will Deacon Cc: Yang Shi --- arch/arm64/include/asm/percpu.h | 70 +++++++++++++++++++++++++++------ 1 file changed, 57 insertions(+), 13 deletions(-) diff --git a/arch/arm64/include/asm/percpu.h b/arch/arm64/include/asm/percpu.h index f19a82437b3fd..5cd98e20e825d 100644 --- a/arch/arm64/include/asm/percpu.h +++ b/arch/arm64/include/asm/percpu.h @@ -490,19 +490,63 @@ PERCPU_CMPXCHG_OP(x, , 64) #define this_cpu_cmpxchg64(pcp, o, n) this_cpu_cmpxchg_8(pcp, o, n) -#define this_cpu_cmpxchg128(pcp, o, n) \ -({ \ - typedef typeof(pcp) pcp_op_T__; \ - u128 old__, new__, ret__; \ - pcp_op_T__ *ptr__; \ - old__ = o; \ - new__ = n; \ - preempt_disable_notrace(); \ - ptr__ = raw_cpu_ptr(&(pcp)); \ - ret__ = cmpxchg128_local((void *)ptr__, old__, new__); \ - preempt_enable_notrace(); \ - ret__; \ -}) +static inline u128 +__percpu_cmpxchg_128(void __percpu *pcp, u128 old, u128 new) +{ + u16 *gprs = ¤t_thread_info()->pcpu_gprs; + unsigned long addr; + unsigned long off; + union __u128_halves r, o = { .full = (old) }, + n = { .full = (new) }; + register unsigned long ol asm ("x0") = o.low; + register unsigned long oh asm ("x1") = o.high; + register unsigned long nl asm ("x2") = n.low; + register unsigned long nh asm ("x3") = n.high; + unsigned long rl, rh; + unsigned long tmp; + + asm volatile ( + __PCPU_GPRS_BEGIN("%[gprs]", "%[pcp]", "%[off]", "%[addr]") + ARM64_LSE_ATOMIC_INSN( + /* LL/SC */ + " prfm pstl1strm, [%[addr]]\n" + "1: ldxp %[rl], %[rh], [%[addr]]\n" + " cmp %[rl], %[ol]\n" + " ccmp %[rh], %[oh], 0, eq\n" + " b.ne 2f\n" + " stxp %w[tmp], %[nl], %[nh], [%[addr]]\n" + " cbnz %w[tmp], 1b\n" + "2:\n" + , + /* LSE atomics */ + " casp %[ol], %[oh], %[nl], %[nh], [%[addr]]\n" + " mov %[rl], %[ol]\n" + " mov %[rh], %[oh]\n" + __nops(4) + ) + __PCPU_GPRS_END("%[gprs]") + : [gprs] "=Qo" (*gprs), + [addr] "=&r" (addr), + [off] "=&r" (off), + [tmp] "=&r" (tmp), + [ol] "+&r" (ol), + [oh] "+&r" (oh), + [rl] "=&r" (rl), + [rh] "=&r" (rh) + : [pcp] "r" (pcp), + [nl] "r" (nl), + [nh] "r" (nh) + : "memory", "cc" + ); + + r.low = rl; + r.high = rh; + + return r.full; +} + +#define this_cpu_cmpxchg128(pcp, o, n) \ + _pcp_wrap_return(__percpu_cmpxchg_128, pcp, o, n) #ifdef __KVM_NVHE_HYPERVISOR__ extern unsigned long __hyp_per_cpu_offset(unsigned int cpu); -- 2.30.2