Linux-ARM-Kernel Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: Mark Rutland <mark.rutland@arm.com>
To: linux-arm-kernel@lists.infradead.org
Cc: mark.rutland@arm.com, vladimir.murzin@arm.com,
	ryan.roberts@arm.com, peterz@infradead.org,
	catalin.marinas@arm.com, david.laight.linux@gmail.com,
	stable@vger.kernel.org, ruanjinjie@huawei.com,
	james.morse@arm.com, yang@os.amperecomputing.com, cl@gentwo.org,
	maz@kernel.org, david@kernel.org, ljs@kernel.org,
	will@kernel.org, ardb@kernel.org
Subject: [PATCH v2 02/20] arm64: percpu: Fix this_cpu_and() mask generation
Date: Tue,  4 Aug 2026 18:04:45 +0100	[thread overview]
Message-ID: <20260804170503.3513916-3-mark.rutland@arm.com> (raw)
In-Reply-To: <20260804170503.3513916-1-mark.rutland@arm.com>

The arm64 implementation of this_cpu_and(pcp, val) is built in terms of
ANDNOT operations, which requires the 'val' argument to be bitwise
negated. The bitwise negation is not implemented correctly, with two
bugs described below.

(1) The bitwise negation is performed as '~val' rather than '~(val)'.
    This won't always generate the expected value when 'val' is an
    expression.

    For example, for this_cpu_and(pcp, 1 - 1):

    * 'val'    is  '1 - 1'   ===> (int) 0x00000000
    * '~val'   is '~1 - 1'   ===> (int) 0xfffffffd
    * '~(val)' is '~(1 - 1)' ===> (int) 0xffffffff

    ... and thus bit[1] of 'pcp' would be preserved unexpectedly by the
    ANDNOT operation.

(2) The bitwise negation is performed on 'val' before it has been cast
    to (at least) the width of 'pcp'. This won't always generate the
    expected value for the upper bits.

    For example, for this_cpu_and(pcp, zero), where 'pcp' is a u64 and
    'zero' is a u32:

    * 'zero'           ===> (u32) 0x00000000
    * '~(zero)'        ===> (u32) 0xffffffff
    * '(u64)~(zero)'   ===> (u64) 0x00000000ffffffff
    * '~((u64)(zero))' ===> (u64) 0xffffffffffffffff

    ... and thus bits[63:32] of 'pcp' would be preserved unexpectedly by
    the ANDNOT operation.

Fix these issues by adding brackets around 'val', and by casting 'val'
to an appropriately-sized type before bitwise negation.

The bugs described above can be seen from the disassembly of the
following test code:

| void this_cpu_and_u64_zero(u64 __percpu *pcp)
| {
|         u64 zero = 0;
|         this_cpu_and(*pcp, zero);
| }
|
| void this_cpu_and_u32_zero(u64 __percpu *pcp)
| {
|         u32 zero = 0;
|         this_cpu_and(*pcp, zero);
| }
|
| void this_cpu_and_expr_zero(u64 __percpu *pcp)
| {
|         this_cpu_and(*pcp, 1 - 1);
| }
|
| void this_cpu_and_expr_zero_brackets(u64 __percpu *pcp)
| {
|         this_cpu_and(*pcp, (1 - 1));
| }

Before this patch:

| <this_cpu_and_u64_zero>:
|        paciasp
|        stp     x29, x30, [sp, #-16]!
|        mrs     x1, sp_el0
|        mov     x29, sp
|        ldr     w2, [x1, #8]
|        add     w2, w2, #0x1
|        str     w2, [x1, #8]
|        mov     x3, #0xffffffffffffffff         // #-1
|        mrs     x2, tpidr_el1
|        add     x0, x0, x2
| 1:     ldxr    x5, [x0]
|        bic     x5, x5, x3
|        stxr    w4, x5, [x0]
|        cbnz    w4, 1b
|        ldr     x0, [x1, #8]
|        add     x0, x0, x3
|        str     w0, [x1, #8]
|        cbz     x0, 2f
|        ldr     x0, [x1, #8]
|        cbnz    x0, 3f
| 2:     bl      preempt_schedule_notrace
| 3:     ldp     x29, x30, [sp], #16
|        autiasp
|        ret
|
| <this_cpu_and_u32_zero>:
|        paciasp
|        stp     x29, x30, [sp, #-16]!
|        mrs     x1, sp_el0
|        mov     x29, sp
|        ldr     w2, [x1, #8]
|        add     w2, w2, #0x1
|        str     w2, [x1, #8]
|        mov     x3, #0xffffffff                 // #4294967295
|        mrs     x2, tpidr_el1
|        add     x0, x0, x2
| 1:     ldxr    x5, [x0]
|        bic     x5, x5, x3
|        stxr    w4, x5, [x0]
|        cbnz    w4, 1b
|        ldr     x0, [x1, #8]
|        sub     x0, x0, #0x1
|        str     w0, [x1, #8]
|        cbz     x0, 2f
|        ldr     x0, [x1, #8]
|        cbnz    x0, 3f
| 2:     bl      preempt_schedule_notrace
| 3:     ldp     x29, x30, [sp], #16
|        autiasp
|        ret
|
| <this_cpu_and_expr_zero>:
|        paciasp
|        stp     x29, x30, [sp, #-16]!
|        mrs     x1, sp_el0
|        mov     x29, sp
|        ldr     w2, [x1, #8]
|        add     w2, w2, #0x1
|        str     w2, [x1, #8]
|        mov     x3, #0xfffffffffffffffd         // #-3
|        mrs     x2, tpidr_el1
|        add     x0, x0, x2
| 1:     ldxr    x5, [x0]
|        bic     x5, x5, x3
|        stxr    w4, x5, [x0]
|        cbnz    w4, 1b
|        ldr     x0, [x1, #8]
|        sub     x0, x0, #0x1
|        str     w0, [x1, #8]
|        cbz     x0, 2f
|        ldr     x0, [x1, #8]
|        cbnz    x0, 3f
| 2:     bl      preempt_schedule_notrace
| 3:     ldp     x29, x30, [sp], #16
|        autiasp
|        ret

After this patch:

| <this_cpu_and_u64_zero>:
|        paciasp
|        stp     x29, x30, [sp, #-16]!
|        mrs     x1, sp_el0
|        mov     x29, sp
|        ldr     w2, [x1, #8]
|        add     w2, w2, #0x1
|        str     w2, [x1, #8]
|        mov     x3, #0xffffffffffffffff         // #-1
|        mrs     x2, tpidr_el1
|        add     x0, x0, x2
| 1:     ldxr    x5, [x0]
|        bic     x5, x5, x3
|        stxr    w4, x5, [x0]
|        cbnz    w4, 1b
|        ldr     x0, [x1, #8]
|        add     x0, x0, x3
|        str     w0, [x1, #8]
|        cbz     x0, 2f
|        ldr     x0, [x1, #8]
|        cbnz    x0, 3f
| 2:     bl      0 <preempt_schedule_notrace>
| 3:     ldp     x29, x30, [sp], #16
|        autiasp
|        ret
|
| <this_cpu_and_u32_zero>:
|        b       this_cpu_and_u64_zero
|
| <this_cpu_and_expr_zero>:
|        b       this_cpu_and_u64_zero
|
| <this_cpu_and_expr_zero_brackets>:
|        b       this_cpu_and_u64_zero

Fixes: 959bf2fd03b5 ("arm64: percpu: Rewrite per-cpu ops to allow use of LSE atomics")
Signed-off-by: Mark Rutland <mark.rutland@arm.com>
Cc: Ada Couprie Diaz <ada.coupriediaz@arm.com>
Cc: Ard Biesheuvel <ardb@kernel.org>
Cc: Catalin Marinas <catalin.marinas@arm.com>
Cc: James Morse <james.morse@arm.com>
Cc: Jinjie Ruan <ruanjinjie@huawei.com>
Cc: Marc Zyngier <maz@kernel.org>
Cc: Peter Zijlstra <peterz@infradead.org>
Cc: Vladimir Murzin <vladimir.murzin@arm.com>
Cc: Will Deacon <will@kernel.org>
Cc: Yang Shi <yang@os.amperecomputing.com>
Cc: stable@vger.kernel.org
---
 arch/arm64/include/asm/percpu.h | 8 ++++----
 1 file changed, 4 insertions(+), 4 deletions(-)

diff --git a/arch/arm64/include/asm/percpu.h b/arch/arm64/include/asm/percpu.h
index 63bbfd4944a37..31193bcf89a2b 100644
--- a/arch/arm64/include/asm/percpu.h
+++ b/arch/arm64/include/asm/percpu.h
@@ -206,13 +206,13 @@ PERCPU_RET_OP(add, add, ldadd)
 	_pcp_protect_return(__percpu_add_return_case_64, pcp, val)
 
 #define this_cpu_and_1(pcp, val)	\
-	_pcp_protect(__percpu_andnot_case_8, pcp, ~val)
+	_pcp_protect(__percpu_andnot_case_8, pcp, ~(u8)(val))
 #define this_cpu_and_2(pcp, val)	\
-	_pcp_protect(__percpu_andnot_case_16, pcp, ~val)
+	_pcp_protect(__percpu_andnot_case_16, pcp, ~(u16)(val))
 #define this_cpu_and_4(pcp, val)	\
-	_pcp_protect(__percpu_andnot_case_32, pcp, ~val)
+	_pcp_protect(__percpu_andnot_case_32, pcp, ~(u32)(val))
 #define this_cpu_and_8(pcp, val)	\
-	_pcp_protect(__percpu_andnot_case_64, pcp, ~val)
+	_pcp_protect(__percpu_andnot_case_64, pcp, ~(u64)(val))
 
 #define this_cpu_or_1(pcp, val)		\
 	_pcp_protect(__percpu_or_case_8, pcp, val)
-- 
2.30.2



  parent reply	other threads:[~2026-08-04 17:05 UTC|newest]

Thread overview: 39+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-08-04 17:04 [PATCH v2 00/20] arm64: Preemptible this_cpu_*() operations Mark Rutland
2026-08-04 17:04 ` [PATCH v2 01/20] arm64: percpu: Fix this_cpu_write() casting Mark Rutland
2026-08-05  8:37   ` David Laight
2026-08-04 17:04 ` Mark Rutland [this message]
2026-08-05  9:14   ` [PATCH v2 02/20] arm64: percpu: Fix this_cpu_and() mask generation David Laight
2026-08-05 13:02     ` Mark Rutland
2026-08-06  8:28       ` David Laight
2026-08-06 10:23         ` Mark Rutland
2026-08-04 17:04 ` [PATCH v2 03/20] arm64: cmpxchg: LL/SC: Avoid redundant extension Mark Rutland
2026-08-04 17:04 ` [PATCH v2 04/20] arm64: cmpxchg128: LSE: Remove redundant operands Mark Rutland
2026-08-04 17:04 ` [PATCH v2 05/20] arm64: preempt: Simplify and optimize __preempt_count_dec_and_test() Mark Rutland
2026-08-04 17:04 ` [PATCH v2 06/20] arm64: preempt: Treat should_resched() as unlikely Mark Rutland
2026-08-04 17:04 ` [PATCH v2 07/20] arm64: ptrace: Always inline pt_regs_[read,write}_reg() Mark Rutland
2026-08-04 17:04 ` [PATCH v2 08/20] arm64: percpu: Factor out percpu offset asm Mark Rutland
2026-08-04 17:04 ` [PATCH v2 09/20] arm64: gpr-num: Add wxN aliases for wN registers Mark Rutland
2026-08-04 17:04 ` [PATCH v2 10/20] arm64: gpr-num: add __GPR_NUM() helper Mark Rutland
2026-08-04 17:04 ` [PATCH v2 11/20] arm64: entry: sdei: Restore all clobberable GPRs Mark Rutland
2026-08-04 17:04 ` [PATCH v2 12/20] arm64: entry: sdei: Make 'tsk' available Mark Rutland
2026-08-04 17:04 ` [PATCH v2 13/20] arm64: percpu: Add infrastructure for preemptible this_cpu_*() ops Mark Rutland
2026-08-04 22:45   ` Pedro Falcato
2026-08-05 10:27     ` David Laight
2026-08-05 12:50       ` Pedro Falcato
2026-08-05  6:45   ` David Hildenbrand (Arm)
2026-08-05  6:47     ` David Hildenbrand (Arm)
2026-08-06 11:21       ` Mark Rutland
2026-08-06 11:32         ` David Hildenbrand (Arm)
2026-08-06 12:02           ` Mark Rutland
2026-08-06 13:25           ` David Laight
2026-08-06 13:30             ` David Hildenbrand (Arm)
2026-08-04 17:04 ` [PATCH v2 14/20] arm64: percpu: Implement preemptible read/write ops Mark Rutland
2026-08-05  9:24   ` Ryan Roberts
2026-08-05 12:08     ` David Laight
2026-08-05 13:34     ` Mark Rutland
2026-08-04 17:04 ` [PATCH v2 15/20] arm64: percpu: Implement preemptible void RMW ops Mark Rutland
2026-08-04 17:04 ` [PATCH v2 16/20] arm64: percpu: Implement preemptible return " Mark Rutland
2026-08-04 17:05 ` [PATCH v2 17/20] arm64: percpu: Implement preemptible XCHG ops Mark Rutland
2026-08-04 17:05 ` [PATCH v2 18/20] arm64: percpu: Implement preemptible CMPXCHG ops Mark Rutland
2026-08-04 17:05 ` [PATCH v2 19/20] arm64: percpu: Implement preemptible CMPXCHG128 ops Mark Rutland
2026-08-04 17:05 ` [PATCH v2 20/20] arm64: percpu: Remove _pcp_protect*() wrappers Mark Rutland

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260804170503.3513916-3-mark.rutland@arm.com \
    --to=mark.rutland@arm.com \
    --cc=ardb@kernel.org \
    --cc=catalin.marinas@arm.com \
    --cc=cl@gentwo.org \
    --cc=david.laight.linux@gmail.com \
    --cc=david@kernel.org \
    --cc=james.morse@arm.com \
    --cc=linux-arm-kernel@lists.infradead.org \
    --cc=ljs@kernel.org \
    --cc=maz@kernel.org \
    --cc=peterz@infradead.org \
    --cc=ruanjinjie@huawei.com \
    --cc=ryan.roberts@arm.com \
    --cc=stable@vger.kernel.org \
    --cc=vladimir.murzin@arm.com \
    --cc=will@kernel.org \
    --cc=yang@os.amperecomputing.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox