From: Vladimir Murzin <vladimir.murzin@arm.com>
To: Mark Rutland <mark.rutland@arm.com>,
linux-arm-kernel@lists.infradead.org
Cc: ryan.roberts@arm.com, usama.anjum@arm.com, peterz@infradead.org,
catalin.marinas@arm.com, david.laight.linux@gmail.com,
stable@vger.kernel.org, ruanjinjie@huawei.com,
james.morse@arm.com, yang@os.amperecomputing.com, cl@gentwo.org,
maz@kernel.org, david@kernel.org, ljs@kernel.org,
will@kernel.org, ardb@kernel.org
Subject: Re: [PATCH v4 19/21] arm64: percpu: Implement preemptible CMPXCHG ops
Date: Tue, 15 Sep 2026 15:20:09 +0100 [thread overview]
Message-ID: <60469883-465f-46dd-a1e1-9d2d746d826a@arm.com> (raw)
In-Reply-To: <20260908151741.394589-20-mark.rutland@arm.com>
On 9/8/26 16:17, Mark Rutland wrote:
> Use the PCPU GPR infrastructure to implement all of the {8,16,32,64}-bit
> cmpxchg ops.
>
> Note that before this patch, the LL/SC and LSE implementation was chosen
> with an alternative branch. After this patch the implementations are
> patched inline, matching the style of the other percpu ops.
>
> Test case:
>
> | u64 outline_this_cpu_cmpxchg_u64(u64 __percpu *p, u64 o, u64 n)
> | {
> | return this_cpu_cmpxchg(*p, o, n);
> | }
>
> Generated code before this patch (v7.2-rc4):
>
> | <outline_this_cpu_cmpxchg_u64>:
> | paciasp
> | stp x29, x30, [sp, #-32]!
> | mrs x4, sp_el0
> | mov x29, sp
> | ldr w3, [x4, #8]
> | add w3, w3, #0x1
> | str w3, [x4, #8]
> | mrs x3, tpidr_el1
> | add x3, x0, x3
> | b 4f // alternative branch
> | cas x1, x2, [x3]
> | mov x0, x1
> | 1: mrs x2, sp_el0
> | ldr x1, [x2, #8]
> | sub x1, x1, #0x1
> | str w1, [x2, #8]
> | cbz x1, 2f
> | ldr x1, [x2, #8]
> | cbnz x1, 3f
> | 2: str x0, [sp, #24]
> | bl preempt_schedule_notrace
> | ldr x0, [sp, #24]
> | 3: ldp x29, x30, [sp], #32
> | autiasp
> | ret
> | 4: prfm pstl1strm, [x3]
> | 5: ldxr x0, [x3]
> | eor x4, x0, x1
> | cbnz x4, 6f
> | stxr w4, x2, [x3]
> | cbnz w4, 5b
> | 6: b 1b
>
> Generated code after this patch:
>
> | <outline_this_cpu_cmpxchg_u64>:
> | mrs x4, sp_el0
> | mov w6, #0x14c0
> | strh w6, [x4, #20]
> | mrs x6, tpidr_el1
> | add x5, x0, x6
> | prfm pstl1strm, [x5]
> | 1: ldxr x3, [x5]
> | eor x7, x3, x1
> | cbnz x7, 2f
> | stxr w7, x2, [x5]
> | cbnz w7, 1b
> | 2: strh wzr, [x4, #20]
> | mov x0, x3
> | ret
>
> Signed-off-by: Mark Rutland <mark.rutland@arm.com>
> Tested-by: Muhammad Usama Anjum <usama.anjum@arm.com>
> Cc: Ada Couprie Diaz <ada.coupriediaz@arm.com>
> Cc: Ard Biesheuvel <ardb@kernel.org>
> Cc: Catalin Marinas <catalin.marinas@arm.com>
> Cc: James Morse <james.morse@arm.com>
> Cc: Jinjie Ruan <ruanjinjie@huawei.com>
> Cc: Marc Zyngier <maz@kernel.org>
> Cc: Peter Zijlstra <peterz@infradead.org>
> Cc: Vladimir Murzin <vladimir.murzin@arm.com>
> Cc: Will Deacon <will@kernel.org>
> Cc: Yang Shi <yang@os.amperecomputing.com>
> ---
> arch/arm64/include/asm/percpu.h | 69 +++++++++++++++++++++++++++++++--
> 1 file changed, 65 insertions(+), 4 deletions(-)
>
> diff --git a/arch/arm64/include/asm/percpu.h b/arch/arm64/include/asm/percpu.h
> index 820340c61d70f..f19a82437b3fd 100644
> --- a/arch/arm64/include/asm/percpu.h
> +++ b/arch/arm64/include/asm/percpu.h
> @@ -306,12 +306,67 @@ __percpu_xchg_case_##sz(void __percpu *pcp, u##sz val) \
> return ret; \
> }
>
> +#define PERCPU_CMPXCHG_OP(w, sfx, sz) \
> +static inline unsigned long \
> +__percpu_cmpxchg_case_##sz(void __percpu *pcp, \
> + u##sz old, \
> + u##sz new) \
> +{ \
> + /* \
> + * Sub-word sizes require zero extension so that EOR+CBNZ don't \
> + * consume non-zero upper bits of the register containing "old".\
> + */ \
> + xwreg_t(w) cmpval = xwreg_zero_extend(old, w, sz); \
> + u16 *gprs = ¤t_thread_info()->pcpu_gprs; \
> + unsigned long addr; \
> + unsigned long off; \
> + unsigned long tmp; \
> + unsigned long oldval; \
> + \
> + asm volatile ( \
> + __PCPU_GPRS_BEGIN("%[gprs]", "%[pcp]", "%[off]", "%[addr]") \
> + ARM64_LSE_ATOMIC_INSN( \
> + /* LL/SC */ \
> + " prfm pstl1strm, [%[addr]]\n" \
> + "1: ldxr" #sfx "\t%" #w "[oldval], [%[addr]]\n" \
> + " eor %" #w "[tmp], %" #w "[oldval], %" #w "[old]\n" \
> + " cbnz %" #w "[tmp], 2f\n" \
> + " stxr" #sfx "\t%w[tmp], %" #w "[new], [%[addr]]\n" \
> + " cbnz %w[tmp], 1b\n" \
> + "2:\n" \
> + , \
> + /* LSE atomics */ \
> + " mov %" #w "[oldval], %" #w "[old]\n" \
> + " cas" #sfx "\t%" #w "[oldval], %" #w "[new], [%[addr]]\n"\
> + __nops(4) \
> + ) \
> + __PCPU_GPRS_END("%[gprs]") \
> + : [gprs] "=Qo" (*gprs), \
> + [addr] "=&r" (addr), \
> + [off] "=&r" (off), \
> + [tmp] "=&r" (tmp), \
> + [oldval] "=&r" (oldval) \
> + : [pcp] "r" (pcp), \
> + [old] "r" (cmpval), \
> + [new] "r" (new) \
> + : "memory" \
> + ); \
> + \
> + return oldval; \
> +}
> +
> PERCPU_XCHG_OP(w, b, 8)
> PERCPU_XCHG_OP(w, h, 16)
> PERCPU_XCHG_OP(w, , 32)
> PERCPU_XCHG_OP(x, , 64)
>
> +PERCPU_CMPXCHG_OP(w, b, 8)
> +PERCPU_CMPXCHG_OP(w, h, 16)
> +PERCPU_CMPXCHG_OP(w, , 32)
> +PERCPU_CMPXCHG_OP(x, , 64)
> +
> #undef PERCPU_XCHG_OP
> +#undef PERCPU_CMPXCHG_OP
>
> /*
> * It would be nice to avoid the conditional call into the scheduler when
> @@ -355,6 +410,12 @@ PERCPU_XCHG_OP(x, , 64)
> (typeof(pcp))op(&(pcp), (unsigned long)(val)); \
> })
>
> +#define _pcp_wrap_cmpxchg(op, pcp, old, new) \
> +({ \
> + (typeof(pcp))op(&(pcp), (unsigned long)(old), \
> + (unsigned long)(new)); \
> +})
> +
> #define this_cpu_read_1(pcp) \
> _pcp_wrap_return(__percpu_read_8, pcp)
> #define this_cpu_read_2(pcp) \
> @@ -419,13 +480,13 @@ PERCPU_XCHG_OP(x, , 64)
> _pcp_wrap_xchg(__percpu_xchg_case_64, pcp, val)
>
> #define this_cpu_cmpxchg_1(pcp, o, n) \
> - _pcp_protect_return(cmpxchg_relaxed, pcp, o, n)
> + _pcp_wrap_cmpxchg(__percpu_cmpxchg_case_8, pcp, o, n)
> #define this_cpu_cmpxchg_2(pcp, o, n) \
> - _pcp_protect_return(cmpxchg_relaxed, pcp, o, n)
> + _pcp_wrap_cmpxchg(__percpu_cmpxchg_case_16, pcp, o, n)
> #define this_cpu_cmpxchg_4(pcp, o, n) \
> - _pcp_protect_return(cmpxchg_relaxed, pcp, o, n)
> + _pcp_wrap_cmpxchg(__percpu_cmpxchg_case_32, pcp, o, n)
> #define this_cpu_cmpxchg_8(pcp, o, n) \
> - _pcp_protect_return(cmpxchg_relaxed, pcp, o, n)
> + _pcp_wrap_cmpxchg(__percpu_cmpxchg_case_64, pcp, o, n)
>
> #define this_cpu_cmpxchg64(pcp, o, n) this_cpu_cmpxchg_8(pcp, o, n)
>
> -- 2.30.2
>
FWIW,
Reviewed-by: Vladimir Murzin <vladimir.murzin@arm.com>
next prev parent reply other threads:[~2026-09-15 14:20 UTC|newest]
Thread overview: 44+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-08 15:17 [PATCH v4 00/21] Preemptible this_cpu_*() operations Mark Rutland
2026-09-08 15:17 ` [PATCH v4 01/21] arm64: percpu: Fix this_cpu_write() casting Mark Rutland
2026-09-10 10:56 ` Lorenzo Stoakes (ARM)
2026-09-08 15:17 ` [PATCH v4 02/21] arm64: percpu: Fix this_cpu_and() mask generation Mark Rutland
2026-09-08 15:17 ` [PATCH v4 03/21] arm64: percpu: Fix LSE operations on {8,16}-bit types Mark Rutland
2026-09-10 15:18 ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 04/21] arm64: cmpxchg: LL/SC: Avoid redundant extension Mark Rutland
2026-09-15 14:21 ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 05/21] arm64: cmpxchg128: LSE: Remove redundant operands Mark Rutland
2026-09-10 10:06 ` Vladimir Murzin
2026-09-11 11:43 ` Mark Rutland
2026-09-08 15:17 ` [PATCH v4 06/21] arm64: preempt: Simplify and optimize __preempt_count_dec_and_test() Mark Rutland
2026-09-08 15:17 ` [PATCH v4 07/21] arm64: preempt: Treat should_resched() as unlikely Mark Rutland
2026-09-08 15:17 ` [PATCH v4 08/21] arm64: ptrace: Always inline pt_regs_[read,write}_reg() Mark Rutland
2026-09-11 10:20 ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 09/21] arm64: percpu: Factor out percpu offset asm Mark Rutland
2026-09-11 12:39 ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 10/21] arm64: gpr-num: Add wxN aliases for wN registers Mark Rutland
2026-09-11 12:43 ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 11/21] arm64: gpr-num: add __GPR_NUM() helper Mark Rutland
2026-09-10 14:24 ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 12/21] arm64: entry: sdei: Restore all clobberable GPRs Mark Rutland
2026-09-11 12:47 ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 13/21] arm64: entry: sdei: Make 'tsk' available Mark Rutland
2026-09-10 13:06 ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 14/21] arm64: percpu: Add infrastructure for preemptible this_cpu_*() ops Mark Rutland
2026-09-15 14:19 ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 15/21] arm64: percpu: Implement preemptible read/write ops Mark Rutland
2026-09-15 11:58 ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 16/21] arm64: percpu: Implement preemptible void RMW ops Mark Rutland
2026-09-15 12:01 ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 17/21] arm64: percpu: Implement preemptible return " Mark Rutland
2026-09-15 12:03 ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 18/21] arm64: percpu: Implement preemptible XCHG ops Mark Rutland
2026-09-15 12:07 ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 19/21] arm64: percpu: Implement preemptible CMPXCHG ops Mark Rutland
2026-09-15 14:20 ` Vladimir Murzin [this message]
2026-09-08 15:17 ` [PATCH v4 20/21] arm64: percpu: Implement preemptible CMPXCHG128 ops Mark Rutland
2026-09-15 14:20 ` Vladimir Murzin
2026-09-08 15:17 ` [PATCH v4 21/21] arm64: percpu: Remove _pcp_protect*() wrappers Mark Rutland
2026-09-15 14:21 ` Vladimir Murzin
2026-09-11 13:44 ` [PATCH v4 00/21] Preemptible this_cpu_*() operations Will Deacon
2026-09-24 16:03 ` (subset) " Catalin Marinas
2026-09-30 13:14 ` Muhammad Usama Anjum
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=60469883-465f-46dd-a1e1-9d2d746d826a@arm.com \
--to=vladimir.murzin@arm.com \
--cc=ardb@kernel.org \
--cc=catalin.marinas@arm.com \
--cc=cl@gentwo.org \
--cc=david.laight.linux@gmail.com \
--cc=david@kernel.org \
--cc=james.morse@arm.com \
--cc=linux-arm-kernel@lists.infradead.org \
--cc=ljs@kernel.org \
--cc=mark.rutland@arm.com \
--cc=maz@kernel.org \
--cc=peterz@infradead.org \
--cc=ruanjinjie@huawei.com \
--cc=ryan.roberts@arm.com \
--cc=stable@vger.kernel.org \
--cc=usama.anjum@arm.com \
--cc=will@kernel.org \
--cc=yang@os.amperecomputing.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.