AMD-GFX Archive on lore.kernel.org
 help / color / mirror / Atom feed
From: "Timur Kristóf" <timur.kristof@gmail.com>
To: natalie.vock@gmx.de, honghuan@amd.com, Alexander.Deucher@amd.com,
	Felix.Kuehling@amd.com, Philip.Yang@amd.com, cascardo@igalia.com,
	tvrtko.ursulin@igalia.com, christian.koenig@amd.com
Cc: amd-gfx@lists.freedesktop.org
Subject: Re: [PATCH 4/9] drm/amdgpu: add amdgpu_vm_pt_leaves() v2
Date: Mon, 28 Sep 2026 15:07:15 -0400	[thread overview]
Message-ID: <0LalfKr0TrmjCRha1xVOMw@gmail.com> (raw)
In-Reply-To: <20260928151041.1857-4-christian.koenig@amd.com>

On 2026. szeptember 28., hétfő 11:10:36 keleti államokbeli nyári idő Christian 
König wrote:
> Add a new function amdgpu_vm_update_leaves() to avoid memory allocation
> on page faults.
> 
> The idea is to only update the leave PDEs/PTEs to let them point to the
> dummy page.
> 
> v2: fix of by one, rework the function to work correctly on PTB as well.
> 
> Signed-off-by: Christian König <christian.koenig@amd.com>
> ---
>  drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c        | 54 ++++++++++----
>  drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h        |  5 +-
>  .../gpu/drm/amd/amdgpu/amdgpu_vm_internal.h   |  3 +
>  drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c     | 74 +++++++++++++++++++
>  4 files changed, 119 insertions(+), 17 deletions(-)
> 
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c
> b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c index f02a99b753c22..d6358cbfcc9b1
> 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c
> @@ -3078,10 +3078,12 @@ bool amdgpu_vm_handle_fault(struct amdgpu_device
> *adev, u32 pasid, u32 vmid, u32 node_id, uint64_t addr,
>  			    uint64_t ts, bool write_fault)
>  {
> +	struct amdgpu_vm_update_params params;
>  	bool is_compute_context = false;
> -	struct drm_exec exec;
> -	uint64_t value, flags;
> +	uint64_t *dst, flags[AMDGPU_VM_MAX_LEVEL];
>  	struct amdgpu_vm *vm;
> +	struct drm_exec exec;
> +	unsigned int idx;
>  	int r;
> 
>  	drm_exec_init(&exec, 0, 1);
> @@ -3125,24 +3127,24 @@ bool amdgpu_vm_handle_fault(struct amdgpu_device
> *adev, u32 pasid, }
> 
>  	addr /= AMDGPU_GPU_PAGE_SIZE;
> -	flags = adev->gmc.init_pte_flags |
> -		AMDGPU_PTE_VALID | AMDGPU_PTE_SNOOPED |
> -		AMDGPU_PTE_SYSTEM;
> -
>  	if (is_compute_context) {
>  		/* Intentionally setting invalid PTE flag
>  		 * combination to force a no-retry-fault
>  		 */
> -		flags = AMDGPU_VM_NORETRY_FLAGS;
> -		value = 0;
> +		for (int i = 0; i < AMDGPU_VM_MAX_LEVEL; ++i)
> +			flags[i] = AMDGPU_VM_NORETRY_FLAGS;
> +		dst = NULL;
>  	} else if (amdgpu_vm_fault_stop == AMDGPU_VM_FAULT_STOP_NEVER) {
>  		/* Redirect the access to the dummy page */
> -		value = adev->dummy_page_addr;
> -		flags |= AMDGPU_PTE_EXECUTABLE | AMDGPU_PTE_READABLE |
> -			 AMDGPU_PTE_WRITEABLE;
> +		for (int i = 0; i < AMDGPU_VM_MAX_LEVEL; ++i)
> +			flags[i] = 0;
> +		dst = adev->vm_manager.dummy_dst;
>  	} else {
>  		/* Let the hw retry silently on the PTE */
> -		value = 0;
> +		for (int i = 0; i < AMDGPU_VM_MAX_LEVEL; ++i)
> +			flags[i] = AMDGPU_PTE_VALID | 
AMDGPU_PTE_SNOOPED |
> +				AMDGPU_PTE_SYSTEM;
> +		dst = NULL;
>  	}
> 
>  	r = dma_resv_reserve_fences(vm->root.bo->tbo.base.resv, 1);
> @@ -3151,12 +3153,32 @@ bool amdgpu_vm_handle_fault(struct amdgpu_device
> *adev, u32 pasid, goto error_unlock;
>  	}
> 
> -	r = amdgpu_vm_update_range(adev, vm, true, false, false, false,
> -				   NULL, addr, addr, flags, value, 
0, NULL, NULL, NULL);
> -	if (r)
> +	if (!drm_dev_enter(adev_to_drm(adev), &idx)) {
> +		r = -ENODEV;
>  		goto error_unlock;
> +	}
> +
> +	memset(&params, 0, sizeof(params));
> +	params.adev = adev;
> +	params.vm = vm;
> +	params.immediate = true;
> +
> +	r = amdgpu_vm_begin_critical(&params);
> +	if (r)
> +		goto error_end_critical;
> +
> +	r = vm->update_funcs->prepare(&params, NULL,
> +				      
AMDGPU_KERNEL_JOB_ID_VM_UPDATE_PDES);
> +	if (r)
> +		goto error_end_critical;
> 
> -	r = amdgpu_vm_update_pdes(adev, vm, true);
> +	amdgpu_vm_pt_leaves(&params, addr, addr + 1, dst, flags);
> +
> +	r = vm->update_funcs->commit(&params, &vm->last_update);

We shouldn't overwrite vm->last_update here.
You can just pass NULL here for now.

> +
> +error_end_critical:
> +	amdgpu_vm_end_critical(&params);
> +	drm_dev_exit(idx);
> 
>  error_unlock:
>  	drm_exec_fini(&exec);
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h
> b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h index c59647554b416..ec5cd38fe4e34
> 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h
> @@ -195,7 +195,10 @@ enum amdgpu_vm_level {
>  	AMDGPU_VM_PDB2,
>  	AMDGPU_VM_PDB1,
>  	AMDGPU_VM_PDB0,
> -	AMDGPU_VM_PTB
> +	AMDGPU_VM_PTB,
> +
> +	/* Not HW level, but for array sizing */
> +	AMDGPU_VM_MAX_LEVEL

Instead of AMDGPU_VM_MAX_LEVEL this should be called AMDGPU_VM_NUM_LEVELS

>  };
> 
>  /* base structure for tracking BO usage in a VM */
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h
> b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h index
> 3c48a3401e2a4..dafdb3a001b8e 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h
> @@ -126,6 +126,9 @@ int amdgpu_vm_pde_update(struct amdgpu_vm_update_params
> *params, int amdgpu_vm_ptes_update(struct amdgpu_vm_update_params *params,
> uint64_t start, uint64_t end,
>  			  uint64_t dst, uint64_t flags);
> +void amdgpu_vm_pt_leaves(struct amdgpu_vm_update_params *params,
> +			 uint64_t start, uint64_t end,
> +			 int64_t *dst, uint64_t *flags);
>  void amdgpu_vm_pt_free_work(struct work_struct *work);
>  void amdgpu_vm_pt_free_list(struct amdgpu_device *adev,
>  			    struct amdgpu_vm_update_params *params);
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c
> b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c index
> 285f17c7705b4..27003b03b5fdb 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c
> @@ -963,6 +963,80 @@ int amdgpu_vm_ptes_update(struct
> amdgpu_vm_update_params *params, return 0;
>  }
> 
> +/**
> + * amdgpu_vm_pt_leaves - update leaf PDEs/PTEs
> + *
> + * @params: see amdgpu_vm_update_params definition
> + * @start: start of GPU address range
> + * @end: end of GPU address range
> + * @dst: optional array with one dst addr per layer
> + * @flags: array of mapping flags per layer
> + *
> + * Update the leaf PDEs/PTEs in the range @start - @end without allocating
> or + * freeing page tables.
> + *
> + * Returns:
> + * 0 for success, negative error code for failure.
> + */
> +void amdgpu_vm_pt_leaves(struct amdgpu_vm_update_params *params,
> +			 uint64_t start, uint64_t end,
> +			 int64_t *dst, uint64_t *flags)
> +{
> +	struct amdgpu_device *adev = params->adev;
> +	struct amdgpu_vm_pt_cursor cursor;
> +
> +	amdgpu_vm_pt_start(adev, params->vm, start, &cursor);
> +	while (cursor.pfn < end) {
> +		unsigned int level, shift, mask, nptes;
> +		uint64_t pe_start, entry_start, entry_end, d, f;
> +		struct amdgpu_bo *pt;
> +
> +		/* Walk to the leave entries */
> +		if (amdgpu_vm_pt_descendant(adev, &cursor))
> +			continue;
> +
> +		if (cursor.entry->bo) {
> +			level = cursor.level;
> +			pt = cursor.entry->bo;
> +		} else {
> +			level = cursor.level - 1;
> +			pt = cursor.parent->bo;
> +		}
> +
> +		shift = amdgpu_vm_pt_level_shift(adev, level);
> +		mask = amdgpu_vm_pt_entries_mask(adev, level);
> +
> +		/* Looks good so far, calculate parameters for the 
update */
> +		pe_start = ((cursor.pfn >> shift) & mask) * 8;
> +
> +		entry_start = cursor.pfn;
> +		if (cursor.entry->bo) {
> +			entry_end = ((uint64_t)mask + 1) << shift;
> +			entry_end += cursor.pfn & ~(entry_end - 1);
> +			entry_end = min(entry_end, end);
> +
> +			nptes = (entry_end - cursor.pfn) >> shift;
> +			amdgpu_vm_pt_next(adev, &cursor);
> +		} else {
> +			nptes = 0;
> +			entry_end = cursor.pfn;
> +			do {
> +				nptes += 1;
> +				entry_end += 1 << shift;
> +				amdgpu_vm_pt_next(adev, &cursor);
> +			} while (cursor.parent && cursor.parent->bo 
== pt &&
> +				 cursor.pfn < end && !
cursor.entry->bo);
> +		}
> +
> +		d = dst ? dst[level] : 0;
> +		f = flags[level];
> +		trace_amdgpu_vm_update_ptes(params, entry_start, 
entry_end,
> +					    min(nptes, 32u), d, 
0, f);
> +		params->vm->update_funcs->update(params, 
to_amdgpu_bo_vm(pt),
> +						 pe_start, 
d, nptes, 0, f);
> +	}
> +}
> +
>  /**
>   * amdgpu_vm_pt_map_tables - have bo of root PD cpu accessible
>   * @adev: amdgpu device structure





  reply	other threads:[~2026-09-28 19:07 UTC|newest]

Thread overview: 25+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-28 15:10 [PATCH 1/9] drm/amdgpu: rework eviction lock handling into critical section v3 Christian König
2026-09-28 15:10 ` [PATCH 2/9] drm/amdgpu: fix cleared PDE/PTE flag generation Christian König
2026-09-28 19:08   ` Timur Kristóf
2026-09-30  9:02     ` Christian König
2026-09-30 14:44       ` Kuehling, Felix
2026-09-30 14:59         ` Christian König
2026-09-30 16:12           ` Kuehling, Felix
2026-10-01  6:24             ` Christian König
2026-10-01 13:36             ` Mukul Joshi
2026-10-01 13:49               ` Joshi, Mukul
2026-09-28 15:10 ` [PATCH 3/9] drm/amdgpu: allocate and fill dummy PDs/PTs Christian König
2026-09-28 19:02   ` Timur Kristóf
2026-09-28 15:10 ` [PATCH 4/9] drm/amdgpu: add amdgpu_vm_pt_leaves() v2 Christian König
2026-09-28 19:07   ` Timur Kristóf [this message]
2026-09-28 15:10 ` [PATCH 5/9] drm/amdgpu: drop immediate updates from amdgpu_vm_update_range Christian König
2026-09-28 15:10 ` [PATCH 6/9] drm/amdgpu: drop immediate updates from amdgpu_vm_update_pdes Christian König
2026-09-28 19:09   ` Timur Kristóf
2026-09-28 15:10 ` [PATCH 7/9] drm/amdgpu: split amdgpu_vm_update_range v3 Christian König
2026-09-30 16:30   ` Kuehling, Felix
2026-09-28 15:10 ` [PATCH 8/9] drm/amdgpu: fix the HMM range handling for KFD SVM v2 Christian König
2026-09-30 16:44   ` Kuehling, Felix
2026-10-01 20:30     ` Olivier Kaloudoff
2026-09-28 15:10 ` [PATCH 9/9] drm/amdgpu: use range unmap in amdgpu_vm_clear_freed Christian König
2026-09-28 19:08 ` [PATCH 1/9] drm/amdgpu: rework eviction lock handling into critical section v3 Timur Kristóf
2026-09-29 14:07 ` Huang, Honglei

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=0LalfKr0TrmjCRha1xVOMw@gmail.com \
    --to=timur.kristof@gmail.com \
    --cc=Alexander.Deucher@amd.com \
    --cc=Felix.Kuehling@amd.com \
    --cc=Philip.Yang@amd.com \
    --cc=amd-gfx@lists.freedesktop.org \
    --cc=cascardo@igalia.com \
    --cc=christian.koenig@amd.com \
    --cc=honghuan@amd.com \
    --cc=natalie.vock@gmx.de \
    --cc=tvrtko.ursulin@igalia.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox