From: "Nicholas Piggin" <npiggin@gmail.com>
To: "Jordan Niethe" <jniethe5@gmail.com>, <linuxppc-dev@lists.ozlabs.org>
Subject: Re: [PATCH 17/17] powerpc/qspinlock: provide accounting and options for sleepy locks
Date: Thu, 10 Nov 2022 21:41:29 +1000 [thread overview]
Message-ID: <CO8L6M3ERSIO.1D8E65M4BMIUK@bobo> (raw)
In-Reply-To: <3ee8195c7491c8e4598977713d10873fc95346e1.camel@gmail.com>
On Thu Nov 10, 2022 at 10:44 AM AEST, Jordan Niethe wrote:
> On Thu, 2022-07-28 at 16:31 +1000, Nicholas Piggin wrote:
> [resend as utf-8, not utf-7]
> > Finding the owner or a queued waiter on a lock with a preempted vcpu
> > is indicative of an oversubscribed guest causing the lock to get into
> > trouble. Provide some options to detect this situation and have new
> > CPUs avoid queueing for a longer time (more steal iterations) to
> > minimise the problems caused by vcpu preemption on the queue.
> > ---
> > arch/powerpc/include/asm/qspinlock_types.h | 7 +-
> > arch/powerpc/lib/qspinlock.c | 240 +++++++++++++++++++--
> > 2 files changed, 232 insertions(+), 15 deletions(-)
> >
> > diff --git a/arch/powerpc/include/asm/qspinlock_types.h b/arch/powerpc/include/asm/qspinlock_types.h
> > index 35f9525381e6..4fbcc8a4230b 100644
> > --- a/arch/powerpc/include/asm/qspinlock_types.h
> > +++ b/arch/powerpc/include/asm/qspinlock_types.h
> > @@ -30,7 +30,7 @@ typedef struct qspinlock {
> > *
> > * 0: locked bit
> > * 1-14: lock holder cpu
> > - * 15: unused bit
> > + * 15: lock owner or queuer vcpus observed to be preempted bit
> > * 16: must queue bit
> > * 17-31: tail cpu (+1)
> > */
> > @@ -49,6 +49,11 @@ typedef struct qspinlock {
> > #error "qspinlock does not support such large CONFIG_NR_CPUS"
> > #endif
> >
> > +#define _Q_SLEEPY_OFFSET 15
> > +#define _Q_SLEEPY_BITS 1
> > +#define _Q_SLEEPY_MASK _Q_SET_MASK(SLEEPY_OWNER)
> > +#define _Q_SLEEPY_VAL (1U << _Q_SLEEPY_OFFSET)
> > +
> > #define _Q_MUST_Q_OFFSET 16
> > #define _Q_MUST_Q_BITS 1
> > #define _Q_MUST_Q_MASK _Q_SET_MASK(MUST_Q)
> > diff --git a/arch/powerpc/lib/qspinlock.c b/arch/powerpc/lib/qspinlock.c
> > index 5cfd69931e31..c18133c01450 100644
> > --- a/arch/powerpc/lib/qspinlock.c
> > +++ b/arch/powerpc/lib/qspinlock.c
> > @@ -5,6 +5,7 @@
> > #include <linux/percpu.h>
> > #include <linux/smp.h>
> > #include <linux/topology.h>
> > +#include <linux/sched/clock.h>
> > #include <asm/qspinlock.h>
> > #include <asm/paravirt.h>
> >
> > @@ -36,24 +37,54 @@ static int HEAD_SPINS __read_mostly = (1<<8);
> > static bool pv_yield_owner __read_mostly = true;
> > static bool pv_yield_allow_steal __read_mostly = false;
> > static bool pv_spin_on_preempted_owner __read_mostly = false;
> > +static bool pv_sleepy_lock __read_mostly = true;
> > +static bool pv_sleepy_lock_sticky __read_mostly = false;
>
> The sticky part could potentially be its own patch.
I'll see how that looks.
> > +static u64 pv_sleepy_lock_interval_ns __read_mostly = 0;
> > +static int pv_sleepy_lock_factor __read_mostly = 256;
> > static bool pv_yield_prev __read_mostly = true;
> > static bool pv_yield_propagate_owner __read_mostly = true;
> > static bool pv_prod_head __read_mostly = false;
> >
> > static DEFINE_PER_CPU_ALIGNED(struct qnodes, qnodes);
> > +static DEFINE_PER_CPU_ALIGNED(u64, sleepy_lock_seen_clock);
> >
> > -static __always_inline int get_steal_spins(bool paravirt, bool remote)
> > +static __always_inline bool recently_sleepy(void)
> > +{
>
> Other users of pv_sleepy_lock_interval_ns first check pv_sleepy_lock.
In this case it should be implied, I've added a comment.
>
> > + if (pv_sleepy_lock_interval_ns) {
> > + u64 seen = this_cpu_read(sleepy_lock_seen_clock);
> > +
> > + if (seen) {
> > + u64 delta = sched_clock() - seen;
> > + if (delta < pv_sleepy_lock_interval_ns)
> > + return true;
> > + this_cpu_write(sleepy_lock_seen_clock, 0);
> > + }
> > + }
> > +
> > + return false;
> > +}
> > +
> > +static __always_inline int get_steal_spins(bool paravirt, bool remote, bool sleepy)
>
> It seems like paravirt is implied by sleepy.
>
> > {
> > if (remote) {
> > - return REMOTE_STEAL_SPINS;
> > + if (paravirt && sleepy)
> > + return REMOTE_STEAL_SPINS * pv_sleepy_lock_factor;
> > + else
> > + return REMOTE_STEAL_SPINS;
> > } else {
> > - return STEAL_SPINS;
> > + if (paravirt && sleepy)
> > + return STEAL_SPINS * pv_sleepy_lock_factor;
> > + else
> > + return STEAL_SPINS;
> > }
> > }
>
> I think that separate functions would still be nicer but this could get rid of
> the nesting conditionals like
>
>
> int spins;
> if (remote)
> spins = REMOTE_STEAL_SPINS;
> else
> spins = STEAL_SPINS;
>
> if (sleepy)
> return spins * pv_sleepy_lock_factor;
> return spins;
Yeah it was getting a bit out of hand.
>
> >
> > -static __always_inline int get_head_spins(bool paravirt)
> > +static __always_inline int get_head_spins(bool paravirt, bool sleepy)
> > {
> > - return HEAD_SPINS;
> > + if (paravirt && sleepy)
> > + return HEAD_SPINS * pv_sleepy_lock_factor;
> > + else
> > + return HEAD_SPINS;
> > }
> >
> > static inline u32 encode_tail_cpu(void)
> > @@ -206,6 +237,60 @@ static __always_inline u32 lock_clear_mustq(struct qspinlock *lock)
> > return prev;
> > }
> >
> > +static __always_inline bool lock_try_set_sleepy(struct qspinlock *lock, u32 old)
> > +{
> > + u32 prev;
> > + u32 new = old | _Q_SLEEPY_VAL;
> > +
> > + BUG_ON(!(old & _Q_LOCKED_VAL));
> > + BUG_ON(old & _Q_SLEEPY_VAL);
> > +
> > + asm volatile(
> > +"1: lwarx %0,0,%1 # lock_try_set_sleepy \n"
> > +" cmpw 0,%0,%2 \n"
> > +" bne- 2f \n"
> > +" stwcx. %3,0,%1 \n"
> > +" bne- 1b \n"
> > +"2: \n"
> > + : "=&r" (prev)
> > + : "r" (&lock->val), "r"(old), "r" (new)
> > + : "cr0", "memory");
> > +
> > + if (prev == old)
> > + return true;
> > + return false;
> > +}
> > +
> > +static __always_inline void seen_sleepy_owner(struct qspinlock *lock, u32 val)
> > +{
> > + if (pv_sleepy_lock) {
> > + if (pv_sleepy_lock_interval_ns)
> > + this_cpu_write(sleepy_lock_seen_clock, sched_clock());
> > + if (!(val & _Q_SLEEPY_VAL))
> > + lock_try_set_sleepy(lock, val);
> > + }
> > +}
> > +
> > +static __always_inline void seen_sleepy_lock(void)
> > +{
> > + if (pv_sleepy_lock && pv_sleepy_lock_interval_ns)
> > + this_cpu_write(sleepy_lock_seen_clock, sched_clock());
> > +}
> > +
> > +static __always_inline void seen_sleepy_node(struct qspinlock *lock)
> > +{
>
> If yield_to_prev() was made to take a raw val, that val could be passed to
> seen_sleepy_node() and it would not need to get it by itself.
Yep.
>
> > + if (pv_sleepy_lock) {
> > + u32 val = READ_ONCE(lock->val);
> > +
> > + if (pv_sleepy_lock_interval_ns)
> > + this_cpu_write(sleepy_lock_seen_clock, sched_clock());
> > + if (val & _Q_LOCKED_VAL) {
> > + if (!(val & _Q_SLEEPY_VAL))
> > + lock_try_set_sleepy(lock, val);
> > + }
> > + }
> > +}
> > +
> > static struct qnode *get_tail_qnode(struct qspinlock *lock, u32 val)
> > {
> > int cpu = get_tail_cpu(val);
> > @@ -244,6 +329,7 @@ static __always_inline void __yield_to_locked_owner(struct qspinlock *lock, u32
> >
> > spin_end();
> >
> > + seen_sleepy_owner(lock, val);
> > *preempted = true;
> >
> > /*
> > @@ -307,11 +393,13 @@ static __always_inline void propagate_yield_cpu(struct qnode *node, u32 val, int
> > }
> > }
> >
> > -static __always_inline void yield_to_prev(struct qspinlock *lock, struct qnode *node, int prev_cpu, bool paravirt)
> > +static __always_inline void yield_to_prev(struct qspinlock *lock, struct qnode *node, int prev_cpu, bool paravirt, bool *preempted)
> > {
> > u32 yield_count;
> > int yield_cpu;
> >
> > + *preempted = false;
> > +
> > if (!paravirt)
> > goto relax;
> >
> > @@ -332,6 +420,9 @@ static __always_inline void yield_to_prev(struct qspinlock *lock, struct qnode *
> >
> > spin_end();
> >
> > + *preempted = true;
> > + seen_sleepy_node(lock);
> > +
> > smp_rmb();
> >
> > if (yield_cpu == node->yield_cpu) {
> > @@ -353,6 +444,9 @@ static __always_inline void yield_to_prev(struct qspinlock *lock, struct qnode *
> >
> > spin_end();
> >
> > + *preempted = true;
> > + seen_sleepy_node(lock);
> > +
> > smp_rmb(); /* See yield_to_locked_owner comment */
> >
> > if (!node->locked) {
> > @@ -369,6 +463,9 @@ static __always_inline void yield_to_prev(struct qspinlock *lock, struct qnode *
> >
> > static __always_inline bool try_to_steal_lock(struct qspinlock *lock, bool paravirt)
> > {
> > + bool preempted;
> > + bool seen_preempted = false;
> > + bool sleepy = false;
> > int iters = 0;
> >
> > if (!STEAL_SPINS) {
> > @@ -376,7 +473,6 @@ static __always_inline bool try_to_steal_lock(struct qspinlock *lock, bool parav
> > spin_begin();
> > for (;;) {
> > u32 val = READ_ONCE(lock->val);
> > - bool preempted;
> >
> > if (val & _Q_MUST_Q_VAL)
> > break;
> > @@ -395,7 +491,6 @@ static __always_inline bool try_to_steal_lock(struct qspinlock *lock, bool parav
> > spin_begin();
> > for (;;) {
> > u32 val = READ_ONCE(lock->val);
> > - bool preempted;
> >
> > if (val & _Q_MUST_Q_VAL)
> > break;
> > @@ -408,9 +503,29 @@ static __always_inline bool try_to_steal_lock(struct qspinlock *lock, bool parav
> > continue;
> > }
> >
> > + if (paravirt && pv_sleepy_lock && !sleepy) {
> > + if (!sleepy) {
>
> The enclosing conditional means this would always be true. I think the out conditional should be
> if (paravirt && pv_sleepy_lock)
> otherwise the pv_sleepy_lock_sticky part wouldn't work properly.
Good catch, I think you're right.
>
>
> > + if (val & _Q_SLEEPY_VAL) {
> > + seen_sleepy_lock();
> > + sleepy = true;
> > + } else if (recently_sleepy()) {
> > + sleepy = true;
> > + }
> > +
> > + if (pv_sleepy_lock_sticky && seen_preempted &&
> > + !(val & _Q_SLEEPY_VAL)) {
> > + if (lock_try_set_sleepy(lock, val))
> > + val |= _Q_SLEEPY_VAL;
> > + }
> > +
> > +
> > yield_to_locked_owner(lock, val, paravirt, &preempted);
> > + if (preempted)
> > + seen_preempted = true;
>
> This could belong to the next if statement, there can not be !paravirt && preempted ?
Yep.
Thanks,
Nick
prev parent reply other threads:[~2022-11-10 11:42 UTC|newest]
Thread overview: 78+ messages / expand[flat|nested] mbox.gz Atom feed top
2022-07-28 6:31 [PATCH 00/17] powerpc: alternate queued spinlock implementation Nicholas Piggin
2022-07-28 6:31 ` [PATCH 01/17] powerpc/qspinlock: powerpc qspinlock implementation Nicholas Piggin
2022-08-10 1:52 ` Jordan NIethe
2022-08-10 6:48 ` Christophe Leroy
2022-11-10 0:35 ` Jordan Niethe
2022-11-10 6:37 ` Christophe Leroy
2022-11-10 11:44 ` Nicholas Piggin
2022-11-10 9:09 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 1a/17] powerpc/qspinlock: Prepare qspinlock code Nicholas Piggin
2022-07-28 6:31 ` [PATCH 02/17] powerpc/qspinlock: add mcs queueing for contended waiters Nicholas Piggin
2022-08-10 2:28 ` Jordan NIethe
2022-11-10 0:36 ` Jordan Niethe
2022-11-10 9:21 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 03/17] powerpc/qspinlock: use a half-word store to unlock to avoid larx/stcx Nicholas Piggin
2022-08-10 3:28 ` Jordan Niethe
2022-11-10 0:39 ` Jordan Niethe
2022-11-10 9:25 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 04/17] powerpc/qspinlock: convert atomic operations to assembly Nicholas Piggin
2022-08-10 3:54 ` Jordan Niethe
2022-11-10 0:39 ` Jordan Niethe
2022-11-10 8:36 ` Christophe Leroy
2022-11-10 11:48 ` Nicholas Piggin
2022-11-10 9:40 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 05/17] powerpc/qspinlock: allow new waiters to steal the lock before queueing Nicholas Piggin
2022-08-10 4:31 ` Jordan Niethe
2022-11-10 0:40 ` Jordan Niethe
2022-11-10 10:54 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 06/17] powerpc/qspinlock: theft prevention to control latency Nicholas Piggin
2022-08-10 5:51 ` Jordan Niethe
2022-11-10 0:40 ` Jordan Niethe
2022-11-10 10:57 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 07/17] powerpc/qspinlock: store owner CPU in lock word Nicholas Piggin
2022-08-12 0:50 ` Jordan Niethe
2022-11-10 0:40 ` Jordan Niethe
2022-11-10 10:59 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 08/17] powerpc/qspinlock: paravirt yield to lock owner Nicholas Piggin
2022-08-12 2:01 ` Jordan Niethe
2022-11-10 0:41 ` Jordan Niethe
2022-11-10 11:13 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 09/17] powerpc/qspinlock: implement option to yield to previous node Nicholas Piggin
2022-08-12 2:07 ` Jordan Niethe
2022-11-10 0:41 ` Jordan Niethe
2022-11-10 11:14 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 10/17] powerpc/qspinlock: allow stealing when head of queue yields Nicholas Piggin
2022-08-12 4:06 ` Jordan Niethe
2022-11-10 0:42 ` Jordan Niethe
2022-11-10 11:22 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 11/17] powerpc/qspinlock: allow propagation of yield CPU down the queue Nicholas Piggin
2022-08-12 4:17 ` Jordan Niethe
2022-10-06 17:27 ` Laurent Dufour
2022-11-10 0:42 ` Jordan Niethe
2022-11-10 11:25 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 12/17] powerpc/qspinlock: add ability to prod new queue head CPU Nicholas Piggin
2022-08-12 4:22 ` Jordan Niethe
2022-11-10 0:42 ` Jordan Niethe
2022-11-10 11:32 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 13/17] powerpc/qspinlock: trylock and initial lock attempt may steal Nicholas Piggin
2022-08-12 4:32 ` Jordan Niethe
2022-11-10 0:43 ` Jordan Niethe
2022-11-10 11:35 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 14/17] powerpc/qspinlock: use spin_begin/end API Nicholas Piggin
2022-08-12 4:36 ` Jordan Niethe
2022-11-10 0:43 ` Jordan Niethe
2022-11-10 11:36 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 15/17] powerpc/qspinlock: reduce remote node steal spins Nicholas Piggin
2022-08-12 4:43 ` Jordan Niethe
2022-11-10 0:43 ` Jordan Niethe
2022-11-10 11:37 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 16/17] powerpc/qspinlock: allow indefinite spinning on a preempted owner Nicholas Piggin
2022-08-12 4:49 ` Jordan Niethe
2022-09-22 15:02 ` Laurent Dufour
2022-09-23 8:16 ` Nicholas Piggin
2022-11-10 0:44 ` Jordan Niethe
2022-11-10 11:38 ` Nicholas Piggin
2022-07-28 6:31 ` [PATCH 17/17] powerpc/qspinlock: provide accounting and options for sleepy locks Nicholas Piggin
2022-08-15 1:11 ` Jordan Niethe
2022-11-10 0:44 ` Jordan Niethe
2022-11-10 11:41 ` Nicholas Piggin [this message]
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=CO8L6M3ERSIO.1D8E65M4BMIUK@bobo \
--to=npiggin@gmail.com \
--cc=jniethe5@gmail.com \
--cc=linuxppc-dev@lists.ozlabs.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox