From: "Christian König" <ckoenig.leichtzumerken@gmail.com>
To: Honglei1.Huang@amd.com, timur.kristof@gmail.com,
natalie.vock@gmx.de, amd-gfx@lists.freedesktop.org
Subject: [PATCH 2/2] drm/amdgpu: rework eviction lock handling into critical section v2
Date: Fri, 11 Sep 2026 18:48:01 +0200 [thread overview]
Message-ID: <20260911164801.50175-2-christian.koenig@amd.com> (raw)
In-Reply-To: <20260911164801.50175-1-christian.koenig@amd.com>
Taking the eviction lock is actually just one step which we need to do
in the critical section handling.
Rename the functions to reflect that, use the update parameters instead of the
vm to save the GFP flags.
v2: rebased and reordered
Signed-off-by: Christian König <christian.koenig@amd.com>
---
drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c | 21 +++++++---
drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h | 1 -
.../gpu/drm/amd/amdgpu/amdgpu_vm_internal.h | 41 ++++++++++++++-----
drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c | 35 ++++++++--------
4 files changed, 63 insertions(+), 35 deletions(-)
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c
index 7ced26c9c651b..67dc7da1a02fa 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c
@@ -1169,11 +1169,9 @@ int amdgpu_vm_update_range(struct amdgpu_device *adev, struct amdgpu_vm *vm,
params.override_pte = allow_override && adev->gmc.override_pte;
INIT_LIST_HEAD(¶ms.tlb_flush_waitlist);
- amdgpu_vm_eviction_lock(vm);
- if (vm->evicting) {
- r = -EBUSY;
+ r = amdgpu_vm_begin_critical(¶ms);
+ if (r)
goto error_free;
- }
if (!unlocked && !dma_fence_is_signaled(vm->last_unlocked)) {
struct dma_fence *tmp = dma_fence_get_stub();
@@ -1257,7 +1255,7 @@ int amdgpu_vm_update_range(struct amdgpu_device *adev, struct amdgpu_vm *vm,
error_free:
kfree(tlb_cb);
- amdgpu_vm_eviction_unlock(vm);
+ amdgpu_vm_end_critical(¶ms);
drm_dev_exit(idx);
return r;
}
@@ -3044,6 +3042,7 @@ bool amdgpu_vm_handle_fault(struct amdgpu_device *adev, u32 pasid,
u32 vmid, u32 node_id, uint64_t addr,
uint64_t ts, bool write_fault)
{
+ struct amdgpu_vm_update_params params;
bool is_compute_context = false;
struct drm_exec exec;
uint64_t value, flags;
@@ -3117,6 +3116,15 @@ bool amdgpu_vm_handle_fault(struct amdgpu_device *adev, u32 pasid,
goto error_unlock;
}
+ memset(¶ms, 0, sizeof(params));
+ params.adev = adev;
+ params.vm = vm;
+ params.immediate = true;
+
+ r = amdgpu_vm_begin_critical(¶ms);
+ if (r)
+ goto error_end_critical;
+
r = amdgpu_vm_update_range(adev, vm, true, false, false, false,
NULL, addr, addr, flags, value, 0, NULL, NULL, NULL);
if (r)
@@ -3124,6 +3132,9 @@ bool amdgpu_vm_handle_fault(struct amdgpu_device *adev, u32 pasid,
r = amdgpu_vm_update_pdes(adev, vm, true);
+error_end_critical:
+ amdgpu_vm_end_critical(¶ms);
+
error_unlock:
drm_exec_fini(&exec);
if (r < 0)
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h
index 0f6634b749746..98cdd7e3475fb 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h
@@ -289,7 +289,6 @@ struct amdgpu_vm {
*/
struct mutex eviction_lock;
bool evicting;
- unsigned int saved_flags;
/* Memory statistics for this vm, protected by stats_lock */
spinlock_t stats_lock;
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h
index ca86eaac75235..77f3942d9533b 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h
@@ -93,6 +93,11 @@ struct amdgpu_vm_update_params {
*/
bool override_pte;
+ /**
+ * @saved_flags: Saved flags for GFP reduction.
+ */
+ unsigned int saved_flags;
+
/**
* @tlb_flush_waitlist: temporary storage for BOs until tlb_flush
*/
@@ -126,21 +131,37 @@ void amdgpu_vm_pt_free_list(struct amdgpu_device *adev,
struct amdgpu_vm_update_params *params);
int amdgpu_vm_pt_map_tables(struct amdgpu_device *adev, struct amdgpu_vm *vm);
-/*
- * vm eviction_lock can be taken in MMU notifiers. Make sure no reclaim-FS
- * happens while holding this lock anywhere to prevent deadlocks when
- * an MMU notifier runs in reclaim-FS context.
+/**
+ * amdgpu_vm_begin_critical - start the critical section of the update
+ * @p: The update parameters
+ *
+ * Serialize all updates, check parameters and make sure that memory allocations
+ * don't enter the reclaim path so that we don't deadlock with MMU notifiers.
+ *
+ * Returns:
+ *
+ * 0 on success or a negative error code on failure.
+ * Even on error amdgpu_vm_end_critical() must still be called to clean up!
*/
-static inline void amdgpu_vm_eviction_lock(struct amdgpu_vm *vm)
+static inline int amdgpu_vm_begin_critical(struct amdgpu_vm_update_params *p)
{
- mutex_lock(&vm->eviction_lock);
- vm->saved_flags = memalloc_noreclaim_save();
+ mutex_lock(&p->vm->eviction_lock);
+ p->saved_flags = memalloc_noreclaim_save();
+ if (p->vm->evicting)
+ return -EBUSY;
+ return 0;
}
-static inline void amdgpu_vm_eviction_unlock(struct amdgpu_vm *vm)
+/**
+ * amdgpu_vm_end_critical - end the critical section of the update
+ * @p: The update parameters
+ *
+ * Restore the GFP flags and drop the lock.
+ */
+static inline void amdgpu_vm_end_critical(struct amdgpu_vm_update_params *p)
{
- memalloc_noreclaim_restore(vm->saved_flags);
- mutex_unlock(&vm->eviction_lock);
+ memalloc_noreclaim_restore(p->saved_flags);
+ mutex_unlock(&p->vm->eviction_lock);
}
#endif
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c
index 1cdf2b854f261..3f78202a60585 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c
@@ -505,51 +505,49 @@ int amdgpu_vm_pt_create(struct amdgpu_device *adev, struct amdgpu_vm *vm,
/**
* amdgpu_vm_pt_alloc - Allocate a specific page table
*
- * @adev: amdgpu_device pointer
- * @vm: VM to allocate page tables for
+ * @p: see amdgpu_vm_update_params definition
* @cursor: Which page table to allocate
- * @immediate: use an immediate update
*
* Make sure a specific page table or directory is allocated.
*
* Returns:
- * 1 if page table needed to be allocated, 0 if page table was already
- * allocated, negative errno if an error occurred.
+ *
+ * 0 on success or a negative error code on failure.
*/
-static int amdgpu_vm_pt_alloc(struct amdgpu_device *adev,
- struct amdgpu_vm *vm,
- struct amdgpu_vm_pt_cursor *cursor,
- bool immediate)
+static int amdgpu_vm_pt_alloc(struct amdgpu_vm_update_params *p,
+ struct amdgpu_vm_pt_cursor *cursor)
{
struct amdgpu_vm_bo_base *entry = cursor->entry;
struct amdgpu_bo *pt_bo;
struct amdgpu_bo_vm *pt;
- int r;
+ int r, r2;
if (entry->bo)
return 0;
- amdgpu_vm_eviction_unlock(vm);
- r = amdgpu_vm_pt_create(adev, vm, cursor->level, immediate, &pt,
- vm->root.bo->xcp_id);
- amdgpu_vm_eviction_lock(vm);
+ amdgpu_vm_end_critical(p);
+ r = amdgpu_vm_pt_create(p->adev, p->vm, cursor->level, p->immediate,
+ &pt, p->vm->root.bo->xcp_id);
+ r2 = amdgpu_vm_begin_critical(p);
if (r)
return r;
+ if (r2)
+ return r2;
/* Keep a reference to the root directory to avoid
* freeing them up in the wrong order.
*/
pt_bo = &pt->bo;
pt_bo->parent = amdgpu_bo_ref(cursor->parent->bo);
- amdgpu_vm_bo_base_init(entry, vm, pt_bo);
- r = amdgpu_vm_pt_clear(adev, vm, pt, immediate);
+ amdgpu_vm_bo_base_init(entry, p->vm, pt_bo);
+ r = amdgpu_vm_pt_clear(p->adev, p->vm, pt, p->immediate);
if (r)
goto error_free_pt;
return 0;
error_free_pt:
- if (vm->is_npa)
+ if (p->vm->is_npa)
amdgpu_bo_unpin(pt_bo);
amdgpu_bo_unref(&pt_bo);
return r;
@@ -838,8 +836,7 @@ int amdgpu_vm_ptes_update(struct amdgpu_vm_update_params *params,
/* make sure that the page tables covering the
* address range are actually allocated
*/
- r = amdgpu_vm_pt_alloc(params->adev, params->vm,
- &cursor, params->immediate);
+ r = amdgpu_vm_pt_alloc(params, &cursor);
if (r)
return r;
}
--
2.43.0
next prev parent reply other threads:[~2026-09-11 16:48 UTC|newest]
Thread overview: 6+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-09-11 16:48 [PATCH 1/2] drm/amdgpu: nuke most amdgpu_vm_eviction_(try)lock uses Christian König
2026-09-11 16:48 ` Christian König [this message]
2026-09-17 8:11 ` [PATCH 2/2] drm/amdgpu: rework eviction lock handling into critical section v2 Natalie Vock
2026-09-11 16:54 ` [PATCH 1/2] drm/amdgpu: nuke most amdgpu_vm_eviction_(try)lock uses Christian König
2026-09-14 8:04 ` Huang, Honglei
2026-09-15 10:58 ` Huang, Honglei
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20260911164801.50175-2-christian.koenig@amd.com \
--to=ckoenig.leichtzumerken@gmail.com \
--cc=Honglei1.Huang@amd.com \
--cc=amd-gfx@lists.freedesktop.org \
--cc=christian.koenig@amd.com \
--cc=natalie.vock@gmx.de \
--cc=timur.kristof@gmail.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.