All of lore.kernel.org
 help / color / mirror / Atom feed
From: "Ghimiray, Himal Prasad" <himal.prasad.ghimiray@intel.com>
To: Matthew Brost <matthew.brost@intel.com>,
	<intel-xe@lists.freedesktop.org>
Cc: <dri-devel@lists.freedesktop.org>, <thomas.hellstrom@linux.intel.com>
Subject: Re: [PATCH v2 2/5] drm/xe: Strict migration policy for atomic SVM faults
Date: Mon, 21 Apr 2025 12:09:15 +0530	[thread overview]
Message-ID: <85da7210-3d79-427d-8199-e852cd6a16a4@intel.com> (raw)
In-Reply-To: <20250417041340.479973-3-matthew.brost@intel.com>



On 17-04-2025 09:43, Matthew Brost wrote:
> Mixing GPU and CPU atomics does not work unless a strict migration
> policy of GPU atomics must be device memory. Enforce a policy of must be
> in VRAM with a retry loop of 2 attempts, if retry loop fails abort
> fault.
> 
> v2:
>   - Only retry migration on atomics
>   - Drop alway migrate modparam
> 
> Signed-off-by: Himal Prasad Ghimiray <himal.prasad.ghimiray@intel.com>
> Signed-off-by: Matthew Brost <matthew.brost@intel.com>
> ---
>   drivers/gpu/drm/xe/xe_module.c |  3 --
>   drivers/gpu/drm/xe/xe_module.h |  1 -
>   drivers/gpu/drm/xe/xe_svm.c    | 57 ++++++++++++++++++++++++++--------
>   drivers/gpu/drm/xe/xe_svm.h    |  5 ---
>   4 files changed, 44 insertions(+), 22 deletions(-)
> 
> diff --git a/drivers/gpu/drm/xe/xe_module.c b/drivers/gpu/drm/xe/xe_module.c
> index 05c7d0ae6d83..1c4dfafbcd0b 100644
> --- a/drivers/gpu/drm/xe/xe_module.c
> +++ b/drivers/gpu/drm/xe/xe_module.c
> @@ -33,9 +33,6 @@ struct xe_modparam xe_modparam = {
>   module_param_named(svm_notifier_size, xe_modparam.svm_notifier_size, uint, 0600);
>   MODULE_PARM_DESC(svm_notifier_size, "Set the svm notifier size(in MiB), must be power of 2");
>   
> -module_param_named(always_migrate_to_vram, xe_modparam.always_migrate_to_vram, bool, 0444);
> -MODULE_PARM_DESC(always_migrate_to_vram, "Always migrate to VRAM on GPU fault");
> -
>   module_param_named_unsafe(force_execlist, xe_modparam.force_execlist, bool, 0444);
>   MODULE_PARM_DESC(force_execlist, "Force Execlist submission");
>   
> diff --git a/drivers/gpu/drm/xe/xe_module.h b/drivers/gpu/drm/xe/xe_module.h
> index 84339e509c80..5a3bfea8b7b4 100644
> --- a/drivers/gpu/drm/xe/xe_module.h
> +++ b/drivers/gpu/drm/xe/xe_module.h
> @@ -12,7 +12,6 @@
>   struct xe_modparam {
>   	bool force_execlist;
>   	bool probe_display;
> -	bool always_migrate_to_vram;
>   	u32 force_vram_bar_size;
>   	int guc_log_level;
>   	char *guc_firmware_path;
> diff --git a/drivers/gpu/drm/xe/xe_svm.c b/drivers/gpu/drm/xe/xe_svm.c
> index 56b18a293bbc..1cc41ce7b684 100644
> --- a/drivers/gpu/drm/xe/xe_svm.c
> +++ b/drivers/gpu/drm/xe/xe_svm.c
> @@ -726,6 +726,35 @@ static int xe_svm_alloc_vram(struct xe_vm *vm, struct xe_tile *tile,
>   }
>   #endif
>   
> +static bool supports_4K_migration(struct xe_device *xe)
> +{
> +	if (xe->info.vram_flags & XE_VRAM_FLAGS_NEED64K)
> +		return false;
> +
> +	return true;
> +}
> +
> +static bool xe_svm_range_needs_migrate_to_vram(struct xe_svm_range *range,
> +					       struct xe_vma *vma)
> +{
> +	struct xe_vm *vm = range_to_vm(&range->base);
> +	u64 range_size = xe_svm_range_size(range);
> +
> +	if (!range->base.flags.migrate_devmem)
> +		return false;
> +
> +	if (xe_svm_range_in_vram(range)) {
> +		drm_dbg(&vm->xe->drm, "Range is already in VRAM\n");
> +		return false;
> +	}
> +
> +	if (range_size <= SZ_64K && !supports_4K_migration(vm->xe)) {
> +		drm_dbg(&vm->xe->drm, "Platform doesn't support SZ_4K range migration\n");
> +		return false;
> +	}
> +
> +	return true;
> +}
>   
>   /**
>    * xe_svm_handle_pagefault() - SVM handle page fault
> @@ -750,12 +779,14 @@ int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
>   			IS_ENABLED(CONFIG_DRM_XE_DEVMEM_MIRROR),
>   		.check_pages_threshold = IS_DGFX(vm->xe) &&
>   			IS_ENABLED(CONFIG_DRM_XE_DEVMEM_MIRROR) ? SZ_64K : 0,
> +		.vram_only = atomic,

atomic && is_dgfx.
  >   	};
>   	struct xe_svm_range *range;
>   	struct drm_gpusvm_range *r;
>   	struct drm_exec exec;
>   	struct dma_fence *fence;
>   	struct xe_tile *tile = gt_to_tile(gt);
> +	int migrate_try_count = atomic ? 3 : 1;
>   	ktime_t end = 0;
>   	int err;
>   
> @@ -782,18 +813,21 @@ int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
>   
>   	range_debug(range, "PAGE FAULT");
>   
> -	/* XXX: Add migration policy, for now migrate range once */
> -	if (!range->skip_migrate && range->base.flags.migrate_devmem &&
> -	    xe_svm_range_size(range) >= SZ_64K) {
> -		range->skip_migrate = true;
> -
> +	if (--migrate_try_count >= 0 &&
> +	    xe_svm_range_needs_migrate_to_vram(range, vma)) {
>   		err = xe_svm_alloc_vram(vm, tile, range, &ctx);
>   		if (err) {
> -			drm_dbg(&vm->xe->drm,
> -				"VRAM allocation failed, falling back to "
> -				"retrying fault, asid=%u, errno=%pe\n",
> -				vm->usm.asid, ERR_PTR(err));
> -			goto retry;
> +			if (migrate_try_count || !ctx.vram_only) {
> +				drm_dbg(&vm->xe->drm,
> +					"VRAM allocation failed, falling back to retrying fault, asid=%u, errno=%pe\n",
> +					vm->usm.asid, ERR_PTR(err));
> +				goto retry;
> +			} else {
> +				drm_err(&vm->xe->drm,
> +					"VRAM allocation failed, retry count exceeded, asid=%u, errno=%pe\n",
> +					vm->usm.asid, ERR_PTR(err));
> +				return err;
> +			}
>   		}
>   	}
>   
> @@ -843,9 +877,6 @@ int xe_svm_handle_pagefault(struct xe_vm *vm, struct xe_vma *vma,
>   	}
>   	drm_exec_fini(&exec);
>   
> -	if (xe_modparam.always_migrate_to_vram)
> -		range->skip_migrate = false;
> -
>   	dma_fence_wait(fence, false);
>   	dma_fence_put(fence);
>   
> diff --git a/drivers/gpu/drm/xe/xe_svm.h b/drivers/gpu/drm/xe/xe_svm.h
> index 3d441eb1f7ea..0e1f376a7471 100644
> --- a/drivers/gpu/drm/xe/xe_svm.h
> +++ b/drivers/gpu/drm/xe/xe_svm.h
> @@ -39,11 +39,6 @@ struct xe_svm_range {
>   	 * range. Protected by GPU SVM notifier lock.
>   	 */
>   	u8 tile_invalidated;
> -	/**
> -	 * @skip_migrate: Skip migration to VRAM, protected by GPU fault handler
> -	 * locking.
> -	 */
> -	u8 skip_migrate	:1;
>   };
>   
>   /**


  reply	other threads:[~2025-04-21  6:39 UTC|newest]

Thread overview: 16+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2025-04-17  4:13 [PATCH v2 0/5] Enable SVM atomics in Xe / GPU SVM Matthew Brost
2025-04-17  4:13 ` [PATCH v2 1/5] drm/gpusvm: Introduce vram_only flag for VRAM allocation Matthew Brost
2025-04-17  4:13 ` [PATCH v2 2/5] drm/xe: Strict migration policy for atomic SVM faults Matthew Brost
2025-04-21  6:39   ` Ghimiray, Himal Prasad [this message]
2025-04-22 15:21     ` Matthew Brost
2025-04-17  4:13 ` [PATCH v2 3/5] drm/gpusvm: Add timeslicing support to GPU SVM Matthew Brost
2025-04-17  4:13 ` [PATCH v2 4/5] drm/xe: Timeslice GPU on atomic SVM fault Matthew Brost
2025-04-17  4:13 ` [PATCH v2 5/5] drm/xe: Add atomic_svm_timeslice_ms debugfs entry Matthew Brost
2025-04-17  4:33 ` ✓ CI.Patch_applied: success for Enable SVM atomics in Xe / GPU SVM (rev2) Patchwork
2025-04-17  4:33 ` ✓ CI.checkpatch: " Patchwork
2025-04-17  4:34 ` ✓ CI.KUnit: " Patchwork
2025-04-17  4:43 ` ✓ CI.Build: " Patchwork
2025-04-17  4:45 ` ✓ CI.Hooks: " Patchwork
2025-04-17  4:47 ` ✓ CI.checksparse: " Patchwork
2025-04-17  5:17 ` ✓ Xe.CI.BAT: " Patchwork
2025-04-17 19:12 ` ✗ Xe.CI.Full: failure " Patchwork

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=85da7210-3d79-427d-8199-e852cd6a16a4@intel.com \
    --to=himal.prasad.ghimiray@intel.com \
    --cc=dri-devel@lists.freedesktop.org \
    --cc=intel-xe@lists.freedesktop.org \
    --cc=matthew.brost@intel.com \
    --cc=thomas.hellstrom@linux.intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.