From mboxrd@z Thu Jan 1 00:00:00 1970 Return-Path: X-Spam-Checker-Version: SpamAssassin 3.4.0 (2014-02-07) on aws-us-west-2-korg-lkml-1.web.codeaurora.org Received: from bombadil.infradead.org (bombadil.infradead.org [198.137.202.133]) (using TLSv1.2 with cipher ECDHE-RSA-AES256-GCM-SHA384 (256/256 bits)) (No client certificate requested) by smtp.lore.kernel.org (Postfix) with ESMTPS id A9531C5518F for ; Tue, 4 Aug 2026 17:06:51 +0000 (UTC) DKIM-Signature: v=1; a=rsa-sha256; q=dns/txt; c=relaxed/relaxed; d=lists.infradead.org; s=bombadil.20210309; h=Sender:Cc:List-Subscribe: List-Help:List-Post:List-Archive:List-Unsubscribe:List-Id: Content-Transfer-Encoding:MIME-Version:References:In-Reply-To:Message-Id:Date :Subject:To:From:Reply-To:Content-Type:Content-ID:Content-Description: Resent-Date:Resent-From:Resent-Sender:Resent-To:Resent-Cc:Resent-Message-ID: List-Owner; bh=udXDuEpF7+oc8ZNFoMC1u2th6iiy0EtAvtRZ0JrhoFA=; b=y2Oa4xfy46qMWj h5CKv0kc44ZIGajCAJHaEHU/kthfZjgFv38iY/IdudtkCy9wAVlsw6XcAAc+AvdLVhIzj0j01ZDbh 0/6UlyUIZRE/Bs5AaPfqF/myE2M0ybJ8pmF7FRlcySYrpDs93Lpm4J07H/6067dIDue2vTpMBq8lU NEDZYtUg2/MHmPVJedYqM5TuKUeEeeKddioTLfQMbee0//8HInbjuHlNooOhiKjU2vf5yLDmenavE y9tprPMdCFCBl7W8x65/6AIZ1ljCnTvnB+pO/zmsGzMiEAk683E5l/wgOE/XhQ1lhLYLhv3btQNSh R/zkcxRI1IphLHQYX6dQ==; Received: from localhost ([::1] helo=bombadil.infradead.org) by bombadil.infradead.org with esmtp (Exim 4.99.1 #2 (Red Hat Linux)) id 1wrIbA-00000002R35-17CA; Tue, 04 Aug 2026 17:06:40 +0000 Received: from foss.arm.com ([217.140.110.172]) by bombadil.infradead.org with esmtp (Exim 4.99.1 #2 (Red Hat Linux)) id 1wrIb8-00000002QzX-0Eog for linux-arm-kernel@lists.infradead.org; Tue, 04 Aug 2026 17:06:39 +0000 Received: from usa-sjc-imap-foss1.foss.arm.com (unknown [10.121.207.14]) by usa-sjc-mx-foss1.foss.arm.com (Postfix) with ESMTP id 151521684; Tue, 4 Aug 2026 10:06:33 -0700 (PDT) Received: from lakrids.cambridge.arm.com (usa-sjc-imap-foss1.foss.arm.com [10.121.207.14]) by usa-sjc-imap-foss1.foss.arm.com (Postfix) with ESMTPA id A19813F632; Tue, 4 Aug 2026 10:06:33 -0700 (PDT) DKIM-Signature: v=1; a=rsa-sha256; c=simple/simple; d=arm.com; s=foss; t=1785863197; bh=qDYhh8nOg0hSrsVgSIs6L9SuDPWSo06I2WPSif1v9uQ=; h=From:To:Cc:Subject:Date:In-Reply-To:References:From; b=YXensOfCURzxuASv6M8g34/hdcdKIQABLTI5KwEs3HpORV2Hv6C2vkxiv8XcEMacq NA+1c9maYVoDojTG2B/Cf4+FdpiCeMfhHxYM4YLz1b0RriINZfLiyXSIR+OO8vCI6q b6axCtPPAJhJHVZEJ0T945WaPV08RQL2//EjfnQo= From: Mark Rutland To: linux-arm-kernel@lists.infradead.org Subject: [PATCH v2 19/20] arm64: percpu: Implement preemptible CMPXCHG128 ops Date: Tue, 4 Aug 2026 18:05:02 +0100 Message-Id: <20260804170503.3513916-20-mark.rutland@arm.com> X-Mailer: git-send-email 2.30.2 In-Reply-To: <20260804170503.3513916-1-mark.rutland@arm.com> References: <20260804170503.3513916-1-mark.rutland@arm.com> MIME-Version: 1.0 Content-Transfer-Encoding: 8bit X-CRM114-Version: 20100106-BlameMichelson ( TRE 0.9.0 (BSD) ) MR-646709E3 X-CRM114-CacheID: sfid-20260804_100638_322831_2EA93FA8 X-CRM114-Status: GOOD ( 12.73 ) X-BeenThere: linux-arm-kernel@lists.infradead.org X-Mailman-Version: 2.1.34 Precedence: list List-Id: List-Unsubscribe: , List-Archive: List-Post: List-Help: List-Subscribe: , Cc: mark.rutland@arm.com, vladimir.murzin@arm.com, ryan.roberts@arm.com, peterz@infradead.org, catalin.marinas@arm.com, david.laight.linux@gmail.com, stable@vger.kernel.org, ruanjinjie@huawei.com, james.morse@arm.com, yang@os.amperecomputing.com, cl@gentwo.org, maz@kernel.org, david@kernel.org, ljs@kernel.org, will@kernel.org, ardb@kernel.org Sender: "linux-arm-kernel" Errors-To: linux-arm-kernel-bounces+linux-arm-kernel=archiver.kernel.org@lists.infradead.org Use the PCPU GPR infrastructure to implement this_cpu_cmpxchg128(). Note that before this patch, the LL/SC and LSE implementation was chosen with an alternative branch. After this patch the implementations are patched inline, matching the style of the other percpu ops. The sub-optimal register shuffling before and after this patch occurs because we hard-code specific registers, as LLVM (currently) lacks a way to place a 128-bit type in a compiler-allocated even-odd pair of registers for inline assembly. We should be able to improve that in future, given compiler support. Test case: | u128 outline_this_cpu_cmpxchg128(u128 __percpu *p, u128 old, u128 new) | { | return this_cpu_cmpxchg128(*p, old, new); | } Generated code before this patch (v7.2-rc4): | : | paciasp | stp x29, x30, [sp, #-32]! | mov x6, x0 | mov x0, x2 | mov x29, sp | mov x1, x3 | mrs x2, sp_el0 | ldr w7, [x2, #8] | add w7, w7, #0x1 | str w7, [x2, #8] | mrs x2, tpidr_el1 | add x6, x6, x2 | b 4f | mov x2, x4 | mov x3, x5 | mov x4, x6 | mov x5, x0 | mov x7, x1 | casp x0, x1, x2, x3, [x6] | 1: mrs x3, sp_el0 | ldr x2, [x3, #8] | sub x2, x2, #0x1 | str w2, [x3, #8] | cbz x2, 2f | ldr x2, [x3, #8] | cbnz x2, 3f | 2: stp x0, x1, [sp, #16] | bl preempt_schedule_notrace | ldp x0, x1, [sp, #16] | 3: ldp x29, x30, [sp], #32 | autiasp | ret | 4: prfm pstl1strm, [x6] | 5: ldxp x8, x7, [x6] | cmp x8, x0 | ccmp x7, x3, #0x0, eq | b.ne 6f | stxp w2, x4, x5, [x6] | cbnz w2, 5b | 6: mov x0, x8 | mov x1, x7 | b 1b Generated code after this patch: | : | mov x6, x0 | mov x1, x3 | mov x0, x2 | mov x3, x5 | mrs x7, sp_el0 | mov x2, x4 | mov w5, #0x10a6 | strh w5, [x7, #20] | mrs x5, tpidr_el1 | add x4, x6, x5 | prfm pstl1strm, [x4] | 1: ldxp x9, x10, [x4] | cmp x9, x0 | ccmp x10, x1, #0x0, eq | b.ne 2f | stxp w8, x2, x3, [x4] | cbnz w8, 1b | 2: strh wzr, [x7, #20] | mov x0, x9 | mov x1, x10 | ret Signed-off-by: Mark Rutland Cc: Ada Couprie Diaz Cc: Ard Biesheuvel Cc: Catalin Marinas Cc: James Morse Cc: Jinjie Ruan Cc: Marc Zyngier Cc: Peter Zijlstra Cc: Vladimir Murzin Cc: Will Deacon Cc: Yang Shi --- arch/arm64/include/asm/percpu.h | 70 +++++++++++++++++++++++++++------ 1 file changed, 57 insertions(+), 13 deletions(-) diff --git a/arch/arm64/include/asm/percpu.h b/arch/arm64/include/asm/percpu.h index af1d4ca5c85ef..8087ed6e12241 100644 --- a/arch/arm64/include/asm/percpu.h +++ b/arch/arm64/include/asm/percpu.h @@ -490,19 +490,63 @@ PERCPU_CMPXCHG_OP(x, , 64) #define this_cpu_cmpxchg64(pcp, o, n) this_cpu_cmpxchg_8(pcp, o, n) -#define this_cpu_cmpxchg128(pcp, o, n) \ -({ \ - typedef typeof(pcp) pcp_op_T__; \ - u128 old__, new__, ret__; \ - pcp_op_T__ *ptr__; \ - old__ = o; \ - new__ = n; \ - preempt_disable_notrace(); \ - ptr__ = raw_cpu_ptr(&(pcp)); \ - ret__ = cmpxchg128_local((void *)ptr__, old__, new__); \ - preempt_enable_notrace(); \ - ret__; \ -}) +static inline u128 +__percpu_cmpxchg_128(void __percpu *pcp, u128 old, u128 new) +{ + u16 *gprs = ¤t_thread_info()->pcpu_gprs; + unsigned long addr; + unsigned long off; + union __u128_halves r, o = { .full = (old) }, + n = { .full = (new) }; + register unsigned long ol asm ("x0") = o.low; + register unsigned long oh asm ("x1") = o.high; + register unsigned long nl asm ("x2") = n.low; + register unsigned long nh asm ("x3") = n.high; + unsigned long rl, rh; + unsigned long tmp; + + asm volatile ( + __PCPU_GPRS_BEGIN("%[gprs]", "%[pcp]", "%[off]", "%[addr]") + ARM64_LSE_ATOMIC_INSN( + /* LL/SC */ + " prfm pstl1strm, [%[addr]]\n" + "1: ldxp %[rl], %[rh], [%[addr]]\n" + " cmp %[rl], %[ol]\n" + " ccmp %[rh], %[oh], 0, eq\n" + " b.ne 2f\n" + " stxp %w[tmp], %[nl], %[nh], [%[addr]]\n" + " cbnz %w[tmp], 1b\n" + "2:\n" + , + /* LSE atomics */ + " casp %[ol], %[oh], %[nl], %[nh], [%[addr]]\n" + " mov %[rl], %[ol]\n" + " mov %[rh], %[oh]\n" + __nops(4) + ) + __PCPU_GPRS_END("%[gprs]") + : [gprs] "=Qo" (*gprs), + [addr] "=&r" (addr), + [off] "=&r" (off), + [tmp] "=&r" (tmp), + [ol] "+&r" (ol), + [oh] "+&r" (oh), + [rl] "=&r" (rl), + [rh] "=&r" (rh) + : [pcp] "r" (pcp), + [nl] "r" (nl), + [nh] "r" (nh) + : "memory", "cc" + ); + + r.low = rl; + r.high = rh; + + return r.full; +} + +#define this_cpu_cmpxchg128(pcp, o, n) \ + _pcp_wrap_return(__percpu_cmpxchg_128, pcp, o, n) #ifdef __KVM_NVHE_HYPERVISOR__ extern unsigned long __hyp_per_cpu_offset(unsigned int cpu); -- 2.30.2