AMD-GFX Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: "Timur Kristóf" <timur.kristof@gmail.com>
To: natalie.vock@gmx.de, honghuan@amd.com, Alexander.Deucher@amd.com,
	Felix.Kuehling@amd.com, Philip.Yang@amd.com, cascardo@igalia.com,
	tvrtko.ursulin@igalia.com, christian.koenig@amd.com
Cc: amd-gfx@lists.freedesktop.org
Subject: Re: [PATCH 1/9] drm/amdgpu: rework eviction lock handling into critical section v3
Date: Mon, 28 Sep 2026 15:08:24 -0400	[thread overview]
Message-ID: <XurSVoYuT2KHd65dYXpHfQ@gmail.com> (raw)
In-Reply-To: <20260928151041.1857-1-christian.koenig@amd.com>

On 2026. szeptember 28., hétfő 11:10:33 keleti államokbeli nyári idő Christian 
König wrote:
> Taking the eviction lock is actually just one step which we need to do
> in the critical section handling.
> 
> Rename the functions to reflect that, use the update parameters instead of
> the vm to save the GFP flags.
> 
> v2: rebased and reordered
> v3: fix rebase artefact, fix error handling in amdgpu_vm_pt_alloc
> 
> Signed-off-by: Christian König <christian.koenig@amd.com>
> Reviewed-by: Natalie Vock <nat@pixelcluster.dev>

Reviewed-by: Timur Kristóf <timur.kristof@gmail.com>

> ---
>  drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c        |  8 ++--
>  drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h        |  1 -
>  .../gpu/drm/amd/amdgpu/amdgpu_vm_internal.h   | 43 ++++++++++++++-----
>  drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c     | 40 ++++++++---------
>  4 files changed, 54 insertions(+), 38 deletions(-)
> 
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c
> b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c index 29a66e39f3d60..7b9494375649f
> 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c
> @@ -1170,11 +1170,9 @@ int amdgpu_vm_update_range(struct amdgpu_device
> *adev, struct amdgpu_vm *vm, params.override_pte = allow_override &&
> adev->gmc.override_pte;
>  	INIT_LIST_HEAD(&params.tlb_flush_waitlist);
> 
> -	amdgpu_vm_eviction_lock(vm);
> -	if (vm->evicting) {
> -		r = -EBUSY;
> +	r = amdgpu_vm_begin_critical(&params);
> +	if (r)
>  		goto error_free;
> -	}
> 
>  	if (!unlocked && !dma_fence_is_signaled(vm->last_unlocked)) {
>  		struct dma_fence *tmp = dma_fence_get_stub();
> @@ -1258,7 +1256,7 @@ int amdgpu_vm_update_range(struct amdgpu_device *adev,
> struct amdgpu_vm *vm,
> 
>  error_free:
>  	kfree(tlb_cb);
> -	amdgpu_vm_eviction_unlock(vm);
> +	amdgpu_vm_end_critical(&params);
>  	drm_dev_exit(idx);
>  	return r;
>  }
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h
> b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h index 0f6634b749746..98cdd7e3475fb
> 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h
> @@ -289,7 +289,6 @@ struct amdgpu_vm {
>  	 */
>  	struct mutex		eviction_lock;
>  	bool			evicting;
> -	unsigned int		saved_flags;
> 
>  	/* Memory statistics for this vm, protected by stats_lock */
>  	spinlock_t		stats_lock;
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h
> b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h index
> ca86eaac75235..8ebb0b033291e 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h
> @@ -51,7 +51,7 @@ struct amdgpu_vm_update_params {
>  	struct amdgpu_device *adev;
> 
>  	/**
> -	 * @vm: optional amdgpu_vm we do this update for
> +	 * @vm: amdgpu_vm we do this update for
>  	 */
>  	struct amdgpu_vm *vm;
> 
> @@ -93,6 +93,11 @@ struct amdgpu_vm_update_params {
>  	 */
>  	bool override_pte;
> 
> +	/**
> +	 * @saved_flags: Saved flags for GFP reduction.
> +	 */
> +	unsigned int saved_flags;
> +
>  	/**
>  	 * @tlb_flush_waitlist: temporary storage for BOs until tlb_flush
>  	 */
> @@ -126,21 +131,37 @@ void amdgpu_vm_pt_free_list(struct amdgpu_device
> *adev, struct amdgpu_vm_update_params *params);
>  int amdgpu_vm_pt_map_tables(struct amdgpu_device *adev, struct amdgpu_vm
> *vm);
> 
> -/*
> - * vm eviction_lock can be taken in MMU notifiers. Make sure no reclaim-FS
> - * happens while holding this lock anywhere to prevent deadlocks when
> - * an MMU notifier runs in reclaim-FS context.
> +/**
> + * amdgpu_vm_begin_critical - start the critical section of the update
> + * @p: The update parameters
> + *
> + * Serialize all updates, check parameters and make sure that memory
> allocations + * don't enter the reclaim path so that we don't deadlock with
> MMU notifiers. + *
> + * Returns:
> + *
> + * 0 on success or a negative error code on failure.
> + * Even on error amdgpu_vm_end_critical() must still be called to clean up!
> */
> -static inline void amdgpu_vm_eviction_lock(struct amdgpu_vm *vm)
> +static inline int amdgpu_vm_begin_critical(struct amdgpu_vm_update_params
> *p) {
> -	mutex_lock(&vm->eviction_lock);
> -	vm->saved_flags = memalloc_noreclaim_save();
> +	mutex_lock(&p->vm->eviction_lock);
> +	p->saved_flags = memalloc_noreclaim_save();
> +	if (p->vm->evicting)
> +		return -EBUSY;
> +	return 0;
>  }
> 
> -static inline void amdgpu_vm_eviction_unlock(struct amdgpu_vm *vm)
> +/**
> + * amdgpu_vm_end_critical - end the critical section of the update
> + * @p: The update parameters
> + *
> + * Restore the GFP flags and drop the lock.
> + */
> +static inline void amdgpu_vm_end_critical(struct amdgpu_vm_update_params
> *p) {
> -	memalloc_noreclaim_restore(vm->saved_flags);
> -	mutex_unlock(&vm->eviction_lock);
> +	memalloc_noreclaim_restore(p->saved_flags);
> +	mutex_unlock(&p->vm->eviction_lock);
>  }
> 
>  #endif
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c
> b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c index
> 1cdf2b854f261..c03327f1242d3 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c
> @@ -505,52 +505,51 @@ int amdgpu_vm_pt_create(struct amdgpu_device *adev,
> struct amdgpu_vm *vm, /**
>   * amdgpu_vm_pt_alloc - Allocate a specific page table
>   *
> - * @adev: amdgpu_device pointer
> - * @vm: VM to allocate page tables for
> + * @p: see amdgpu_vm_update_params definition
>   * @cursor: Which page table to allocate
> - * @immediate: use an immediate update
>   *
>   * Make sure a specific page table or directory is allocated.
>   *
>   * Returns:
> - * 1 if page table needed to be allocated, 0 if page table was already
> - * allocated, negative errno if an error occurred.
> + *
> + * 0 on success or a negative error code on failure.
>   */
> -static int amdgpu_vm_pt_alloc(struct amdgpu_device *adev,
> -			      struct amdgpu_vm *vm,
> -			      struct amdgpu_vm_pt_cursor *cursor,
> -			      bool immediate)
> +static int amdgpu_vm_pt_alloc(struct amdgpu_vm_update_params *p,
> +			      struct amdgpu_vm_pt_cursor *cursor)
>  {
>  	struct amdgpu_vm_bo_base *entry = cursor->entry;
>  	struct amdgpu_bo *pt_bo;
>  	struct amdgpu_bo_vm *pt;
> -	int r;
> +	int r, r2;
> 
>  	if (entry->bo)
>  		return 0;
> 
> -	amdgpu_vm_eviction_unlock(vm);
> -	r = amdgpu_vm_pt_create(adev, vm, cursor->level, immediate, &pt,
> -				vm->root.bo->xcp_id);
> -	amdgpu_vm_eviction_lock(vm);
> +	amdgpu_vm_end_critical(p);
> +	r = amdgpu_vm_pt_create(p->adev, p->vm, cursor->level, p-
>immediate,
> +				&pt, p->vm->root.bo->xcp_id);
> +	r2 = amdgpu_vm_begin_critical(p);
>  	if (r)
>  		return r;
> +	if (r2)
> +		goto error_free_pt;
> 
>  	/* Keep a reference to the root directory to avoid
>  	 * freeing them up in the wrong order.
>  	 */
>  	pt_bo = &pt->bo;
>  	pt_bo->parent = amdgpu_bo_ref(cursor->parent->bo);
> -	amdgpu_vm_bo_base_init(entry, vm, pt_bo);
> -	r = amdgpu_vm_pt_clear(adev, vm, pt, immediate);
> +	amdgpu_vm_bo_base_init(entry, p->vm, pt_bo);
> +	r = amdgpu_vm_pt_clear(p->adev, p->vm, pt, p->immediate);
>  	if (r)
> -		goto error_free_pt;
> +		goto error_unpin;
> 
>  	return 0;
> 
> -error_free_pt:
> -	if (vm->is_npa)
> +error_unpin:
> +	if (p->vm->is_npa)
>  		amdgpu_bo_unpin(pt_bo);
> +error_free_pt:
>  	amdgpu_bo_unref(&pt_bo);
>  	return r;
>  }
> @@ -838,8 +837,7 @@ int amdgpu_vm_ptes_update(struct amdgpu_vm_update_params
> *params, /* make sure that the page tables covering the
>  			 * address range are actually allocated
>  			 */
> -			r = amdgpu_vm_pt_alloc(params->adev, params-
>vm,
> -					       &cursor, 
params->immediate);
> +			r = amdgpu_vm_pt_alloc(params, &cursor);
>  			if (r)
>  				return r;
>  		}





  parent reply	other threads:[~2026-09-28 19:08 UTC|newest]

Thread overview: 25+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-28 15:10 [PATCH 1/9] drm/amdgpu: rework eviction lock handling into critical section v3 Christian König
2026-09-28 15:10 ` [PATCH 2/9] drm/amdgpu: fix cleared PDE/PTE flag generation Christian König
2026-09-28 19:08   ` Timur Kristóf
2026-09-30  9:02     ` Christian König
2026-09-30 14:44       ` Kuehling, Felix
2026-09-30 14:59         ` Christian König
2026-09-30 16:12           ` Kuehling, Felix
2026-10-01  6:24             ` Christian König
2026-10-01 13:36             ` Mukul Joshi
2026-10-01 13:49               ` Joshi, Mukul
2026-09-28 15:10 ` [PATCH 3/9] drm/amdgpu: allocate and fill dummy PDs/PTs Christian König
2026-09-28 19:02   ` Timur Kristóf
2026-09-28 15:10 ` [PATCH 4/9] drm/amdgpu: add amdgpu_vm_pt_leaves() v2 Christian König
2026-09-28 19:07   ` Timur Kristóf
2026-09-28 15:10 ` [PATCH 5/9] drm/amdgpu: drop immediate updates from amdgpu_vm_update_range Christian König
2026-09-28 15:10 ` [PATCH 6/9] drm/amdgpu: drop immediate updates from amdgpu_vm_update_pdes Christian König
2026-09-28 19:09   ` Timur Kristóf
2026-09-28 15:10 ` [PATCH 7/9] drm/amdgpu: split amdgpu_vm_update_range v3 Christian König
2026-09-30 16:30   ` Kuehling, Felix
2026-09-28 15:10 ` [PATCH 8/9] drm/amdgpu: fix the HMM range handling for KFD SVM v2 Christian König
2026-09-30 16:44   ` Kuehling, Felix
2026-10-01 20:30     ` Olivier Kaloudoff
2026-09-28 15:10 ` [PATCH 9/9] drm/amdgpu: use range unmap in amdgpu_vm_clear_freed Christian König
2026-09-28 19:08 ` Timur Kristóf [this message]
2026-09-29 14:07 ` [PATCH 1/9] drm/amdgpu: rework eviction lock handling into critical section v3 Huang, Honglei

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=XurSVoYuT2KHd65dYXpHfQ@gmail.com \
    --to=timur.kristof@gmail.com \
    --cc=Alexander.Deucher@amd.com \
    --cc=Felix.Kuehling@amd.com \
    --cc=Philip.Yang@amd.com \
    --cc=amd-gfx@lists.freedesktop.org \
    --cc=cascardo@igalia.com \
    --cc=christian.koenig@amd.com \
    --cc=honghuan@amd.com \
    --cc=natalie.vock@gmx.de \
    --cc=tvrtko.ursulin@igalia.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox