Intel-XE Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: Matthew Brost <matthew.brost@intel.com>
To: Zongyao Bai <zongyao.bai@intel.com>
Cc: <intel-xe@lists.freedesktop.org>, <jia.yao@intel.com>
Subject: Re: [PATCH] drm/xe/forcewake: add delayed-release optimization
Date: Wed, 10 Jun 2026 20:09:33 -0700	[thread overview]
Message-ID: <aiom7bfet3CddhLq@gsse-cloud1.jf.intel.com> (raw)
In-Reply-To: <20260601213804.707256-1-zongyao.bai@intel.com>

On Mon, Jun 01, 2026 at 09:38:04PM +0000, Zongyao Bai wrote:
> Add delayed-release optimization:
>     - Add domain sleep 200us after xe_force_wake_put()
>     - Skip MMIO wake in xe_force_wake_get() if domain still awake.
> Reduces frequent wake/sleep cycles for back-to-back operations.
> Examples of scenarios: zeDeviceGetGlobalTimestamps read by VTune, PTI
> 

I think this concept makes sense, as MMIO read operations are relatively
expensive in terms of time cost (perhaps ~5 µs). However, the downside
is increased power usage.

Should we make delayed release an optional call—for example,
xe_force_wake_put_delay—and only use it in specific critical paths? For
instance, we could limit its use to paths tied to Level Zero calls like
zeDeviceGetGlobalTimestamps.

This gets a bit tricky if xe_force_wake_put_delay is called and is not
the last reference, followed by xe_force_wake_put being the final
reference. However, it should be straightforward to track that an
delayed put was requested and have the final xe_force_wake_put issue the
delay.

> Signed-off-by: Zongyao Bai <zongyao.bai@intel.com>
> ---
>  drivers/gpu/drm/xe/xe_force_wake.c       | 111 +++++++++++++++++------
>  drivers/gpu/drm/xe/xe_force_wake.h       |   4 +-
>  drivers/gpu/drm/xe/xe_force_wake_types.h |  11 +++
>  drivers/gpu/drm/xe/xe_gt.c               |   4 +-
>  4 files changed, 99 insertions(+), 31 deletions(-)
> 
> diff --git a/drivers/gpu/drm/xe/xe_force_wake.c b/drivers/gpu/drm/xe/xe_force_wake.c
> index 197e2197bd0a..183a17fa6d68 100644
> --- a/drivers/gpu/drm/xe/xe_force_wake.c
> +++ b/drivers/gpu/drm/xe/xe_force_wake.c
> @@ -6,15 +6,20 @@
>  #include "xe_force_wake.h"
>  
>  #include <drm/drm_util.h>
> +#include <linux/device.h>
> +#include <linux/hrtimer.h>
>  
>  #include "regs/xe_gt_regs.h"
>  #include "regs/xe_reg_defs.h"
> +#include "xe_device.h"
>  #include "xe_gt.h"
>  #include "xe_gt_printk.h"
>  #include "xe_mmio.h"
> +#include "xe_pm.h"
>  #include "xe_sriov.h"
>  
>  #define XE_FORCE_WAKE_ACK_TIMEOUT_MS	50
> +#define XE_FORCE_WAKE_HOLD_DELAY_US	200

How did you choose this value? It seems like it should be something
configurable via Kconfig or configfs.

>  
>  static const char *str_wake_sleep(bool wake)
>  {
> @@ -27,6 +32,8 @@ static void mark_domain_initialized(struct xe_force_wake *fw,
>  	fw->initialized_domains |= BIT(id);
>  }
>  
> +static enum hrtimer_restart xe_force_wake_domain_timer(struct hrtimer *timer);
> +
>  static void init_domain(struct xe_force_wake *fw,
>  			enum xe_force_wake_domain_id id,
>  			struct xe_reg reg, struct xe_reg ack)
> @@ -38,11 +45,29 @@ static void init_domain(struct xe_force_wake *fw,
>  	domain->reg_ack = ack;
>  	domain->val = FORCEWAKE_MT(FORCEWAKE_KERNEL);
>  	domain->mask = FORCEWAKE_MT_MASK(FORCEWAKE_KERNEL);
> +	domain->fw_back = fw;
> +	hrtimer_setup(&domain->timer, xe_force_wake_domain_timer,
> +		      CLOCK_MONOTONIC, HRTIMER_MODE_REL);
>  
>  	mark_domain_initialized(fw, id);
>  }
>  
> -void xe_force_wake_init_gt(struct xe_gt *gt, struct xe_force_wake *fw)
> +static void xe_force_wake_fini(void *arg)
> +{
> +	struct xe_force_wake *fw = arg;
> +	struct xe_gt *gt = fw->gt;
> +	struct xe_force_wake_domain *domain;
> +	unsigned int tmp;
> +
> +	for_each_fw_domain(domain, fw, tmp) {
> +		xe_gt_WARN(gt, domain->ref,
> +			   "Forcewake domain %d still referenced (%u) at teardown\n",
> +			   domain->id, domain->ref);

Is the warning above actually valid? It seems fairly normal for a fini
to race with a delayed fw put. I would drop this.

> +		hrtimer_cancel(&domain->timer);
> +	}
> +}
> +
> +int xe_force_wake_init_gt(struct xe_gt *gt, struct xe_force_wake *fw)
>  {
>  	struct xe_device *xe = gt_to_xe(gt);
>  
> @@ -58,6 +83,8 @@ void xe_force_wake_init_gt(struct xe_gt *gt, struct xe_force_wake *fw)
>  			    FORCEWAKE_GT,
>  			    FORCEWAKE_ACK_GT);
>  	}
> +
> +	return devm_add_action_or_reset(xe->drm.dev, xe_force_wake_fini, fw);
>  }
>  
>  void xe_force_wake_init_engines(struct xe_gt *gt, struct xe_force_wake *fw)
> @@ -142,10 +169,36 @@ static void domain_sleep(struct xe_gt *gt, struct xe_force_wake_domain *domain)
>  	__domain_ctl(gt, domain, false);
>  }
>  
> -static int domain_sleep_wait(struct xe_gt *gt,
> -			     struct xe_force_wake_domain *domain)
> +static enum hrtimer_restart xe_force_wake_domain_timer(struct hrtimer *timer)
>  {
> -	return __domain_wait(gt, domain, false);
> +	struct xe_force_wake_domain *domain =
> +		container_of(timer, struct xe_force_wake_domain, timer);
> +	struct xe_force_wake *fw = domain->fw_back;
> +	struct xe_gt *gt = fw->gt;
> +	unsigned long flags;
> +
> +	xe_gt_assert(gt, !xe_pm_runtime_suspended(gt_to_xe(gt)));
> +
> +	spin_lock_irqsave(&fw->lock, flags);

I'd use guard(spinlock_irqsave) here rather manually unlock this.
hrtimer_forward_now should be safe under fw->lock unless I'm missing
something.

> +
> +	if (!(fw->timer_domains & BIT(domain->id)) || domain->ref) {
> +		spin_unlock_irqrestore(&fw->lock, flags);
> +		return HRTIMER_NORESTART;
> +	}
> +	if (domain->timer_rearm) {
> +		domain->timer_rearm = false;
> +		spin_unlock_irqrestore(&fw->lock, flags);
> +		hrtimer_forward_now(timer,
> +				    ns_to_ktime(XE_FORCE_WAKE_HOLD_DELAY_US *
> +						NSEC_PER_USEC));
> +		return HRTIMER_RESTART;
> +	}
> +	fw->timer_domains &= ~BIT(domain->id);
> +	domain_sleep(gt, domain);
> +	fw->awake_domains &= ~BIT(domain->id);
> +	spin_unlock_irqrestore(&fw->lock, flags);
> +
> +	return HRTIMER_NORESTART;
>  }
>  
>  /**
> @@ -187,8 +240,13 @@ unsigned int __must_check xe_force_wake_get(struct xe_force_wake *fw,
>  	spin_lock_irqsave(&fw->lock, flags);
>  	for_each_fw_domain_masked(domain, ref_rqst, fw, tmp) {
>  		if (!domain->ref++) {
> -			awake_rqst |= BIT(domain->id);
> -			domain_wake(gt, domain);
> +			if (fw->awake_domains & BIT(domain->id)) {
> +				fw->timer_domains &= ~BIT(domain->id);
> +				hrtimer_try_to_cancel(&domain->timer);
> +			} else {
> +				awake_rqst |= BIT(domain->id);
> +				domain_wake(gt, domain);
> +			}
>  		}
>  		ref_incr |= BIT(domain->id);
>  	}
> @@ -213,27 +271,25 @@ unsigned int __must_check xe_force_wake_get(struct xe_force_wake *fw,
>  }
>  
>  /**
> - * xe_force_wake_put - Decrement the refcount and put domain to sleep if refcount becomes 0
> + * xe_force_wake_put - Decrement the refcount and arm the delayed-sleep timer
>   * @fw: Pointer to the force wake structure
>   * @fw_ref: return of xe_force_wake_get()
>   *
> - * This function reduces the reference counts for domains in fw_ref. If
> - * refcount for any of the specified domain reaches 0, it puts the domain to sleep
> - * and waits for acknowledgment for domain to sleep within 50 milisec timeout.
> - * Warns in case of timeout of ack from domain.
> + * This function reduces the reference counts for domains in fw_ref.  When a
> + * domain's refcount reaches 0 the sleep request is not issued immediately;
> + * instead a hrtimer is armed for XE_FORCE_WAKE_HOLD_DELAY_US so that a rapid
> + * xe_force_wake_get() can reuse the still-awake domain at zero MMIO cost.  On
> + * timer expiry, if the domain is still idle, the sleep request is written.
> + * Mirroring i915's fw_domains_put(), the deferred sleep is fire-and-forget:
> + * no sleep ACK is polled, since the next wake re-waits for the wake ACK.

Let's not mention the i915 in Xe code.

>   */
>  void xe_force_wake_put(struct xe_force_wake *fw, unsigned int fw_ref)
>  {
>  	struct xe_gt *gt = fw->gt;
>  	struct xe_force_wake_domain *domain;
> -	unsigned int tmp, sleep = 0;
> +	unsigned int tmp;
>  	unsigned long flags;
> -	int ack_fail = 0;
>  
> -	/*
> -	 * Avoid unnecessary lock and unlock when the function is called
> -	 * in error path of individual domains.
> -	 */

Why delete this comment?

>  	if (!fw_ref)
>  		return;
>  
> @@ -245,20 +301,19 @@ void xe_force_wake_put(struct xe_force_wake *fw, unsigned int fw_ref)
>  		xe_gt_assert(gt, domain->ref);
>  
>  		if (!--domain->ref) {
> -			sleep |= BIT(domain->id);
> -			domain_sleep(gt, domain);
> +			fw->timer_domains |= BIT(domain->id);
> +			if (hrtimer_callback_running(&domain->timer)) {
> +				domain->timer_rearm = true;
> +			} else {
> +				domain->timer_rearm = false;
> +				hrtimer_start(&domain->timer,
> +					      ns_to_ktime(XE_FORCE_WAKE_HOLD_DELAY_US *
> +							  NSEC_PER_USEC),
> +				      HRTIMER_MODE_REL);
> +			}
>  		}
>  	}
> -	for_each_fw_domain_masked(domain, sleep, fw, tmp) {
> -		if (domain_sleep_wait(gt, domain) == 0)
> -			fw->awake_domains &= ~BIT(domain->id);
> -		else
> -			ack_fail |= BIT(domain->id);
> -	}
>  	spin_unlock_irqrestore(&fw->lock, flags);
> -
> -	xe_gt_WARN(gt, ack_fail, "Forcewake domain%s %#x failed to acknowledge sleep request\n",
> -		   str_plural(hweight_long(ack_fail)), ack_fail);

This deleted code for domain_sleep_wait / error probably needs to be in
xe_force_wake_domain_timer.

>  }
>  
>  const char *xe_force_wake_domain_to_str(enum xe_force_wake_domain_id id)
> diff --git a/drivers/gpu/drm/xe/xe_force_wake.h b/drivers/gpu/drm/xe/xe_force_wake.h
> index e2721f205d6c..19679b923dca 100644
> --- a/drivers/gpu/drm/xe/xe_force_wake.h
> +++ b/drivers/gpu/drm/xe/xe_force_wake.h
> @@ -11,8 +11,8 @@
>  
>  struct xe_gt;
>  
> -void xe_force_wake_init_gt(struct xe_gt *gt,
> -			   struct xe_force_wake *fw);
> +int xe_force_wake_init_gt(struct xe_gt *gt,
> +			  struct xe_force_wake *fw);
>  void xe_force_wake_init_engines(struct xe_gt *gt,
>  				struct xe_force_wake *fw);
>  unsigned int __must_check xe_force_wake_get(struct xe_force_wake *fw,
> diff --git a/drivers/gpu/drm/xe/xe_force_wake_types.h b/drivers/gpu/drm/xe/xe_force_wake_types.h
> index 14b7b86e801b..ee5675069fe0 100644
> --- a/drivers/gpu/drm/xe/xe_force_wake_types.h
> +++ b/drivers/gpu/drm/xe/xe_force_wake_types.h
> @@ -6,6 +6,7 @@
>  #ifndef _XE_FORCE_WAKE_TYPES_H_
>  #define _XE_FORCE_WAKE_TYPES_H_
>  
> +#include <linux/hrtimer.h>
>  #include <linux/mutex.h>
>  #include <linux/types.h>
>  
> @@ -51,6 +52,8 @@ enum xe_force_wake_domains {
>  	XE_FORCEWAKE_ALL	= BIT(XE_FW_DOMAIN_ID_COUNT)
>  };
>  
> +struct xe_force_wake;
> +
>  /**
>   * struct xe_force_wake_domain - Xe force wake power domain
>   *
> @@ -82,6 +85,12 @@ struct xe_force_wake_domain {
>  	u32 mask;
>  	/** @ref: domain reference */
>  	u32 ref;
> +	/** @timer_rearm: put() ran while callback was in-flight; callback must restart timer */

Protected fw_back->lock.

> +	bool timer_rearm;

In general, I’d reorganize the layout so that structs are at the top of
xe_force_wake_domain, followed by u32 fields, and finally the bool
fields.

> +	/** @timer: hrtimer for delayed sleep request */
> +	struct hrtimer timer;
> +	/** @fw_back: back pointer to parent xe_force_wake */
> +	struct xe_force_wake *fw_back;
>  };
>  
>  /**
> @@ -101,6 +110,8 @@ struct xe_force_wake {
>  	spinlock_t lock;
>  	/** @awake_domains: mask of all domains awake */
>  	unsigned int awake_domains;
> +	/** @timer_domains: mask of domains with an outstanding delayed-sleep timer */
> +	unsigned int timer_domains;
>  	/** @initialized_domains: mask of all initialized domains */
>  	unsigned int initialized_domains;
>  	/** @domains: force wake domains */
> diff --git a/drivers/gpu/drm/xe/xe_gt.c b/drivers/gpu/drm/xe/xe_gt.c
> index 783eb6d631b5..43a79698cd04 100644
> --- a/drivers/gpu/drm/xe/xe_gt.c
> +++ b/drivers/gpu/drm/xe/xe_gt.c
> @@ -511,7 +511,9 @@ int xe_gt_init_early(struct xe_gt *gt)
>  
>  	xe_wa_process_gt_oob(gt);
>  
> -	xe_force_wake_init_gt(gt, gt_to_fw(gt));
> +	err = xe_force_wake_init_gt(gt, gt_to_fw(gt));
> +	if (err)
> +		return err;

I'd add newline here.

Matt

>  	spin_lock_init(&gt->global_invl_lock);
>  
>  	err = xe_gt_tlb_inval_init_early(gt);
> -- 
> 2.43.0
> 

  parent reply	other threads:[~2026-06-11  3:09 UTC|newest]

Thread overview: 34+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-06-01 21:38 [PATCH] drm/xe/forcewake: add delayed-release optimization Zongyao Bai
2026-06-01 22:37 ` ✓ CI.KUnit: success for " Patchwork
2026-06-01 23:15 ` ✓ Xe.CI.BAT: " Patchwork
2026-06-02  8:13 ` ✓ Xe.CI.FULL: " Patchwork
2026-06-11  1:03 ` [PATCH v2] " Zongyao Bai
2026-06-11 11:59   ` Maarten Lankhorst
2026-06-18 21:18     ` Bai, Zongyao
2026-06-11  1:13 ` ✓ CI.KUnit: success for drm/xe/forcewake: add delayed-release optimization (rev2) Patchwork
2026-06-11  1:58 ` ✓ Xe.CI.BAT: " Patchwork
2026-06-11  3:09 ` Matthew Brost [this message]
2026-06-11  3:15   ` [PATCH] drm/xe/forcewake: add delayed-release optimization Matthew Brost
2026-06-25  1:37     ` Bai, Zongyao
2026-06-11 11:29 ` ✓ Xe.CI.FULL: success for drm/xe/forcewake: add delayed-release optimization (rev2) Patchwork
2026-06-25  8:19 ` [PATCH v3] drm/xe/forcewake: add delayed-release optimization Zongyao Bai
2026-06-25  8:51 ` ✓ CI.KUnit: success for drm/xe/forcewake: add delayed-release optimization (rev3) Patchwork
2026-06-25  9:26 ` ✗ Xe.CI.BAT: failure " Patchwork
2026-06-25 10:55 ` ✗ Xe.CI.FULL: " Patchwork
2026-07-20 22:13 ` [PATCH v4] drm/xe/forcewake: add delayed-release optimization Zongyao Bai
2026-07-21 22:43   ` Matthew Brost
2026-07-20 22:18 ` ✗ CI.checkpatch: warning for drm/xe/forcewake: add delayed-release optimization (rev4) Patchwork
2026-07-20 22:20 ` ✓ CI.KUnit: success " Patchwork
2026-07-20 22:54 ` ✓ Xe.CI.BAT: " Patchwork
2026-07-21  5:00 ` ✓ Xe.CI.FULL: " Patchwork
2026-08-13  0:06 ` [PATCH v5 0/4] drm/xe/forcewake: add delayed-release optimization Zongyao Bai
2026-08-13  0:06   ` [PATCH v5 1/4] drm/xe/forcewake: synchronize engine-cycle access with unplug Zongyao Bai
2026-08-13  0:21     ` sashiko-bot
2026-08-13  0:06   ` [PATCH v5 2/4] drm/xe/forcewake: add delayed-release state machine Zongyao Bai
2026-08-13  0:23     ` sashiko-bot
2026-08-13  0:06   ` [PATCH v5 3/4] drm/xe/forcewake: flush delayed release at power boundaries Zongyao Bai
2026-08-13  0:23     ` sashiko-bot
2026-08-13  0:06   ` [PATCH v5 4/4] drm/xe/forcewake: enable configurable delayed forcewake release Zongyao Bai
2026-08-13  0:18     ` sashiko-bot
2026-08-13  0:13 ` ✗ CI.checkpatch: warning for drm/xe/forcewake: add delayed-release optimization (rev5) Patchwork
2026-08-13  0:15 ` ✓ CI.KUnit: success " Patchwork

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=aiom7bfet3CddhLq@gsse-cloud1.jf.intel.com \
    --to=matthew.brost@intel.com \
    --cc=intel-xe@lists.freedesktop.org \
    --cc=jia.yao@intel.com \
    --cc=zongyao.bai@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox