dri-devel Archive on lore.kernel.org
 help / color / mirror / Atom feed
* [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback
@ 2026-09-30  6:21 Arunpravin Paneer Selvam
  2026-09-30  6:31 ` sashiko-bot
                   ` (2 more replies)
  0 siblings, 3 replies; 8+ messages in thread
From: Arunpravin Paneer Selvam @ 2026-09-30  6:21 UTC (permalink / raw)
  To: matthew.auld, dri-devel, intel-gfx, intel-xe, amd-gfx
  Cc: christian.koenig, alexander.deucher, Anand.Raghavendra,
	Arunpravin Paneer Selvam

From: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com>

A range + contiguous allocation (e.g. a scanout FB confined to the
CPU-visible VRAM aperture) rounds its size up to a power of two and
requires a naturally aligned free block of that size; on a fragmented
aperture no such aligned block may exist even though enough contiguous
space is free, so the allocation fails with -ENOSPC.

The non-range contiguous path already recovers from this via
__alloc_contig_try_harder(), which stitches an exact-size span from
smaller adjacent blocks, but that fallback was unreachable once
GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder()
a [range_start, range_end) window and route the range + contiguous
case through it: each candidate placement is confined to the window
and aligned to min_block_size. Non-range callers pass [0, mm->size),
where the guards are no-ops, so the existing behaviour is unchanged.
A KUnit regression test covers the fragmentation pattern.

Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b
regression.

v2:
 - Drop the split-undo patch; the range-bias search always descends to
   an exact-order block or fails the split (already handled), so the
   extra undo was redundant. (Matthew)
 - Verified this fallback alone fixes the kms_plane regression.

Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with decoupled dirty tracker")
Assisted-by: Claude:claude-opus-4-8
Cc: Matthew Auld <matthew.auld@intel.com>
Cc: Christian König <christian.koenig@amd.com>
Signed-off-by: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com>
---
 drivers/gpu/buddy.c                | 88 ++++++++++++++++--------------
 drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++--
 2 files changed, 126 insertions(+), 45 deletions(-)

diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c
index 2f2aaadafe35..5265e1f6a318 100644
--- a/drivers/gpu/buddy.c
+++ b/drivers/gpu/buddy.c
@@ -1685,26 +1685,14 @@ static int __gpu_buddy_alloc_range(struct gpu_buddy *mm,
 			     blocks, total_allocated_on_err);
 }
 
-static int __alloc_contig_aligned_retry(struct gpu_buddy *mm,
-					u64 unaligned_offset,
-					u64 size,
-					u64 min_block_size,
-					unsigned long flags,
-					struct list_head *blocks)
-{
-	u64 aligned_offset = round_down(unaligned_offset, min_block_size);
-
-	return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags,
-				       NULL, blocks);
-}
-
 static int __alloc_contig_try_harder(struct gpu_buddy *mm,
+				     u64 range_start, u64 range_end,
 				     u64 size,
 				     u64 min_block_size,
 				     unsigned long flags,
 				     struct list_head *blocks)
 {
-	u64 rhs_offset, lhs_offset, filled;
+	u64 rhs_offset, lhs_offset, filled, aligned;
 	struct gpu_buddy_block *block;
 	struct rb_root *root;
 	struct rb_node *iter;
@@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct gpu_buddy *mm,
 					       flags, &filled, blocks);
 		if (err && err != -ENOSPC)
 			return err;
-		if (!err && IS_ALIGNED(rhs_offset, min_block_size))
+		if (!err && IS_ALIGNED(rhs_offset, min_block_size) &&
+		    rhs_offset >= range_start && rhs_offset + size <= range_end)
 			return 0;
 		if (!err) {
 			/* Allocate the unaligned RHS offset using round_down */
 			gpu_buddy_free_list_internal(mm, blocks);
-			err = __alloc_contig_aligned_retry(mm, rhs_offset,
-							   size,
-							   min_block_size,
-							   flags, blocks);
-			if (!err)
-				return 0;
-			if (err != -ENOSPC) {
-				gpu_buddy_free_list_internal(mm, blocks);
-				return err;
+
+			aligned = round_down(rhs_offset, min_block_size);
+			if (aligned >= range_start &&
+			    aligned + size <= range_end) {
+				err = __gpu_buddy_alloc_range(mm, aligned, size,
+							      flags, NULL, blocks);
+				if (!err)
+					return 0;
+				if (err != -ENOSPC) {
+					gpu_buddy_free_list_internal(mm, blocks);
+					return err;
+				}
 			}
 			goto next;
 		}
@@ -1759,15 +1751,17 @@ static int __alloc_contig_try_harder(struct gpu_buddy *mm,
 
 		/* Allocate the unaligned LHS offset using round_down */
 		gpu_buddy_free_list_internal(mm, blocks);
-		err = __alloc_contig_aligned_retry(mm, lhs_offset,
-						   size,
-						   min_block_size,
-						   flags, blocks);
-		if (!err)
-			return 0;
-		if (err != -ENOSPC) {
-			gpu_buddy_free_list_internal(mm, blocks);
-			return err;
+
+		aligned = round_down(lhs_offset, min_block_size);
+		if (aligned >= range_start && aligned + size <= range_end) {
+			err = __gpu_buddy_alloc_range(mm, aligned, size,
+						      flags, NULL, blocks);
+			if (!err)
+				return 0;
+			if (err != -ENOSPC) {
+				gpu_buddy_free_list_internal(mm, blocks);
+				return err;
+			}
 		}
 next:
 		gpu_buddy_free_list_internal(mm, blocks);
@@ -2021,11 +2015,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
 	min_order = ilog2(min_block_size) - ilog2(mm->chunk_size);
 
 	if (order > mm->max_order || size > mm->size) {
-		if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) &&
-		    !(flags & GPU_BUDDY_RANGE_ALLOCATION))
-			return __alloc_contig_try_harder(mm, original_size,
+		if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
+			u64 range_start, range_end;
+
+			range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? start : 0;
+			range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : mm->size;
+
+			return __alloc_contig_try_harder(mm, range_start,
+							 range_end,
+							 original_size,
 							 original_min_size,
 							 flags, blocks);
+		}
 
 		return -EINVAL;
 	}
@@ -2058,9 +2059,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
 			 * Try contiguous block allocation through
 			 * try harder method.
 			 */
-			if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION &&
-			    !(flags & GPU_BUDDY_RANGE_ALLOCATION)) {
-				err = __alloc_contig_try_harder(mm,
+			if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
+				u64 range_start, range_end;
+
+				range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? start : 0;
+				range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : mm->size;
+
+				err = __alloc_contig_try_harder(mm, range_start,
+								range_end,
 								original_size,
 								original_min_size,
 								flags,
@@ -2068,9 +2074,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
 				if (!err)
 					return 0;
 				if (err != -ENOSPC)
-					return err;
-				goto err_free;
+					goto err_free;
 			}
+
 			err = -ENOSPC;
 			goto err_free;
 		} while (1);
diff --git a/drivers/gpu/tests/gpu_buddy_test.c b/drivers/gpu/tests/gpu_buddy_test.c
index b75d32ca6ca0..2c445870b808 100644
--- a/drivers/gpu/tests/gpu_buddy_test.c
+++ b/drivers/gpu/tests/gpu_buddy_test.c
@@ -1251,6 +1251,77 @@ static void gpu_test_buddy_alloc_contiguous(struct kunit *test)
 	gpu_buddy_fini(&mm);
 }
 
+static void gpu_test_buddy_alloc_range_contiguous(struct kunit *test)
+{
+	const unsigned long ps = SZ_4K, mm_size = 16 * ps;
+	const unsigned long range_end = 8 * ps;
+	struct gpu_buddy_block *block, *prev;
+	LIST_HEAD(allocated);
+	struct gpu_buddy mm;
+	LIST_HEAD(pin_lo);
+	LIST_HEAD(pin_hi);
+	u64 total;
+
+	KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps),
+			       "buddy_init failed\n");
+
+	/*
+	 * Idea is to confine the test to the sub-range [0, 32K), which a 12K
+	 * contiguous request (rounded up to 16K) splits into two naturally
+	 * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the first 4K page
+	 * of each slot ([0, 4K) and [16K, 20K)) so that neither aligned slot
+	 * can satisfy the rounded-up 16K allocation, yet the freed remainder
+	 * still leaves a contiguous 12K hole at offset 4K, which is page-aligned
+	 * but not 16K-aligned. A 12K contiguous+range allocation must therefore
+	 * fall back to stitching that span instead of returning -ENOSPC.
+	 */
+	KUNIT_ASSERT_FALSE_MSG(test,
+			       gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps,
+						      &pin_lo, 0),
+			       "failed to pin low page\n");
+	KUNIT_ASSERT_FALSE_MSG(test,
+			       gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps,
+						      ps, &pin_hi, 0),
+			       "failed to pin high page\n");
+
+	/* No aligned 16K block is free; the range-aware fallback must stitch
+	 * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC.
+	 */
+	KUNIT_ASSERT_FALSE_MSG(test,
+			       gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps,
+						      ps, &allocated,
+						      GPU_BUDDY_CONTIGUOUS_ALLOCATION |
+						      GPU_BUDDY_RANGE_ALLOCATION),
+			       "range-restricted contiguous alloc failed\n");
+
+	/* The result must be exactly 3*ps, contiguous, and inside the range. */
+	total = 0;
+	prev = NULL;
+	list_for_each_entry(block, &allocated, link) {
+		u64 offset = gpu_buddy_block_offset(block);
+		u64 bsize = gpu_buddy_block_size(&mm, block);
+
+		KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end,
+				      "block [%llx, %llx) outside range\n",
+				      offset, offset + bsize);
+		if (prev)
+			KUNIT_EXPECT_EQ_MSG(test,
+					    gpu_buddy_block_offset(prev) +
+					    gpu_buddy_block_size(&mm, prev),
+					    offset,
+					    "block at %llx not contiguous\n",
+					    offset);
+		prev = block;
+		total += bsize;
+	}
+	KUNIT_EXPECT_EQ(test, total, 3 * ps);
+
+	gpu_buddy_free_list(&mm, &allocated, 0);
+	gpu_buddy_free_list(&mm, &pin_lo, 0);
+	gpu_buddy_free_list(&mm, &pin_hi, 0);
+	gpu_buddy_fini(&mm);
+}
+
 static void gpu_test_buddy_alloc_pathological(struct kunit *test)
 {
 	u64 mm_size, size, start = 0;
@@ -1534,10 +1605,13 @@ static void gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test)
 				     GPU_BUDDY_RANGE_ALLOCATION);
 	KUNIT_EXPECT_EQ(test, err, -EINVAL);
 
-	/* CONTIGUOUS + RANGE should return -EINVAL (no try_harder for RANGE) */
-	err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, &blocks,
-				     GPU_BUDDY_CONTIGUOUS_ALLOCATION | GPU_BUDDY_RANGE_ALLOCATION);
-	KUNIT_EXPECT_EQ(test, err, -EINVAL);
+	/* CONTIGUOUS + RANGE should succeed via the range-aware try_harder */
+	KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, mm_size, size,
+							    SZ_4K, &blocks,
+							    GPU_BUDDY_CONTIGUOUS_ALLOCATION |
+							    GPU_BUDDY_RANGE_ALLOCATION),
+			       "range contiguous alloc hit an error size=%llu\n", size);
+	gpu_buddy_free_list(&mm, &blocks, 0);
 
 	gpu_buddy_fini(&mm);
 }
@@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = {
 	KUNIT_CASE(gpu_test_buddy_alloc_pessimistic),
 	KUNIT_CASE(gpu_test_buddy_alloc_pathological),
 	KUNIT_CASE(gpu_test_buddy_alloc_contiguous),
+	KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous),
 	KUNIT_CASE(gpu_test_buddy_alloc_clear),
 	KUNIT_CASE(gpu_test_buddy_alloc_range),
 	KUNIT_CASE(gpu_test_buddy_alloc_range_bias),

base-commit: 744f262401cc2e1f3827496c72a47e072d31a852
-- 
2.43.0


^ permalink raw reply related	[flat|nested] 8+ messages in thread

* Re: [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback
  2026-09-30  6:21 [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback Arunpravin Paneer Selvam
@ 2026-09-30  6:31 ` sashiko-bot
  2026-10-01  7:06 ` Arunpravin Paneer Selvam
  2026-10-01 18:28 ` Matthew Auld
  2 siblings, 0 replies; 8+ messages in thread
From: sashiko-bot @ 2026-09-30  6:31 UTC (permalink / raw)
  To: Arunpravin Paneer Selvam; +Cc: dri-devel, intel-xe, intel-gfx

Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider:
- [High] gpu/buddy: range-restricted alloc fails when free space exceeds range_end
--

commit 2ed4c9d1a70c51390968b65e939095ce52c8ee07
Author: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com>

gpu/buddy: add range-restricted contiguous allocation fallback

A range + contiguous allocation rounds its size up to a power of two
and requires a naturally aligned free block of that size. The
non-range contiguous path already recovers from fragmentation via
__alloc_contig_try_harder(), which is now given a range window to
support the range + contiguous case.

> diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c
> index 2f2aaadafe351..5265e1f6a3183 100644
> --- a/drivers/gpu/buddy.c
> +++ b/drivers/gpu/buddy.c
[ ... ]
> @@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct gpu_buddy *mm,
[ ... ]
>  		if (!err) {
>  			/* Allocate the unaligned RHS offset using round_down */
>  			gpu_buddy_free_list_internal(mm, blocks);
> -			err = __alloc_contig_aligned_retry(mm, rhs_offset,
> -							   size,
> -							   min_block_size,
> -							   flags, blocks);
> -			if (!err)
> -				return 0;
> -			if (err != -ENOSPC) {
> -				gpu_buddy_free_list_internal(mm, blocks);
> -				return err;
> +
> +			aligned = round_down(rhs_offset, min_block_size);
> +			if (aligned >= range_start &&
> +			    aligned + size <= range_end) {
> +				err = __gpu_buddy_alloc_range(mm, aligned, size,
> +							      flags, NULL, blocks);
> +				if (!err)
> +					return 0;
> +				if (err != -ENOSPC) {
> +					gpu_buddy_free_list_internal(mm, blocks);
> +					return err;
> +				}
>  			}
>  			goto next;
>  		}

[Severity: High]
Does this skip the left-hand side (LHS) fallback when contiguous free space
exceeds range_end?

If __gpu_buddy_alloc_range() succeeds at finding free space starting at
rhs_offset, err is 0. If the free space extends beyond range_end, the newly
added check (aligned >= range_start && aligned + size <= range_end) might
fail. But since err is still 0, it hits the unconditional goto next.

By skipping to next, doesn't this bypass the LHS window-sliding logic below?
If there was valid free space to the left of rhs_offset that perfectly fits
the window, it appears it would be ignored, causing the allocator to return
-ENOSPC despite sufficient space existing within the requested range.

-- 
Sashiko AI review · https://sashiko.dev/#/patchset/20260930062117.585089-1-arunpravin.paneerselvam@amd.com?part=1

^ permalink raw reply	[flat|nested] 8+ messages in thread

* Re: [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback
  2026-09-30  6:21 [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback Arunpravin Paneer Selvam
  2026-09-30  6:31 ` sashiko-bot
@ 2026-10-01  7:06 ` Arunpravin Paneer Selvam
  2026-10-01 18:28 ` Matthew Auld
  2 siblings, 0 replies; 8+ messages in thread
From: Arunpravin Paneer Selvam @ 2026-10-01  7:06 UTC (permalink / raw)
  To: matthew.auld, dri-devel, intel-gfx, intel-xe, amd-gfx
  Cc: christian.koenig, alexander.deucher, Anand.Raghavendra

Hi Matthew,
Could you please review this patch.

Thanks,
Arun.

On 9/30/2026 11:51 AM, Arunpravin Paneer Selvam wrote:
> From: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com>
>
> A range + contiguous allocation (e.g. a scanout FB confined to the
> CPU-visible VRAM aperture) rounds its size up to a power of two and
> requires a naturally aligned free block of that size; on a fragmented
> aperture no such aligned block may exist even though enough contiguous
> space is free, so the allocation fails with -ENOSPC.
>
> The non-range contiguous path already recovers from this via
> __alloc_contig_try_harder(), which stitches an exact-size span from
> smaller adjacent blocks, but that fallback was unreachable once
> GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder()
> a [range_start, range_end) window and route the range + contiguous
> case through it: each candidate placement is confined to the window
> and aligned to min_block_size. Non-range callers pass [0, mm->size),
> where the guards are no-ops, so the existing behaviour is unchanged.
> A KUnit regression test covers the fragmentation pattern.
>
> Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b
> regression.
>
> v2:
>   - Drop the split-undo patch; the range-bias search always descends to
>     an exact-order block or fails the split (already handled), so the
>     extra undo was redundant. (Matthew)
>   - Verified this fallback alone fixes the kms_plane regression.
>
> Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with decoupled dirty tracker")
> Assisted-by: Claude:claude-opus-4-8
> Cc: Matthew Auld <matthew.auld@intel.com>
> Cc: Christian König <christian.koenig@amd.com>
> Signed-off-by: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com>
> ---
>   drivers/gpu/buddy.c                | 88 ++++++++++++++++--------------
>   drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++--
>   2 files changed, 126 insertions(+), 45 deletions(-)
>
> diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c
> index 2f2aaadafe35..5265e1f6a318 100644
> --- a/drivers/gpu/buddy.c
> +++ b/drivers/gpu/buddy.c
> @@ -1685,26 +1685,14 @@ static int __gpu_buddy_alloc_range(struct gpu_buddy *mm,
>   			     blocks, total_allocated_on_err);
>   }
>   
> -static int __alloc_contig_aligned_retry(struct gpu_buddy *mm,
> -					u64 unaligned_offset,
> -					u64 size,
> -					u64 min_block_size,
> -					unsigned long flags,
> -					struct list_head *blocks)
> -{
> -	u64 aligned_offset = round_down(unaligned_offset, min_block_size);
> -
> -	return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags,
> -				       NULL, blocks);
> -}
> -
>   static int __alloc_contig_try_harder(struct gpu_buddy *mm,
> +				     u64 range_start, u64 range_end,
>   				     u64 size,
>   				     u64 min_block_size,
>   				     unsigned long flags,
>   				     struct list_head *blocks)
>   {
> -	u64 rhs_offset, lhs_offset, filled;
> +	u64 rhs_offset, lhs_offset, filled, aligned;
>   	struct gpu_buddy_block *block;
>   	struct rb_root *root;
>   	struct rb_node *iter;
> @@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct gpu_buddy *mm,
>   					       flags, &filled, blocks);
>   		if (err && err != -ENOSPC)
>   			return err;
> -		if (!err && IS_ALIGNED(rhs_offset, min_block_size))
> +		if (!err && IS_ALIGNED(rhs_offset, min_block_size) &&
> +		    rhs_offset >= range_start && rhs_offset + size <= range_end)
>   			return 0;
>   		if (!err) {
>   			/* Allocate the unaligned RHS offset using round_down */
>   			gpu_buddy_free_list_internal(mm, blocks);
> -			err = __alloc_contig_aligned_retry(mm, rhs_offset,
> -							   size,
> -							   min_block_size,
> -							   flags, blocks);
> -			if (!err)
> -				return 0;
> -			if (err != -ENOSPC) {
> -				gpu_buddy_free_list_internal(mm, blocks);
> -				return err;
> +
> +			aligned = round_down(rhs_offset, min_block_size);
> +			if (aligned >= range_start &&
> +			    aligned + size <= range_end) {
> +				err = __gpu_buddy_alloc_range(mm, aligned, size,
> +							      flags, NULL, blocks);
> +				if (!err)
> +					return 0;
> +				if (err != -ENOSPC) {
> +					gpu_buddy_free_list_internal(mm, blocks);
> +					return err;
> +				}
>   			}
>   			goto next;
>   		}
> @@ -1759,15 +1751,17 @@ static int __alloc_contig_try_harder(struct gpu_buddy *mm,
>   
>   		/* Allocate the unaligned LHS offset using round_down */
>   		gpu_buddy_free_list_internal(mm, blocks);
> -		err = __alloc_contig_aligned_retry(mm, lhs_offset,
> -						   size,
> -						   min_block_size,
> -						   flags, blocks);
> -		if (!err)
> -			return 0;
> -		if (err != -ENOSPC) {
> -			gpu_buddy_free_list_internal(mm, blocks);
> -			return err;
> +
> +		aligned = round_down(lhs_offset, min_block_size);
> +		if (aligned >= range_start && aligned + size <= range_end) {
> +			err = __gpu_buddy_alloc_range(mm, aligned, size,
> +						      flags, NULL, blocks);
> +			if (!err)
> +				return 0;
> +			if (err != -ENOSPC) {
> +				gpu_buddy_free_list_internal(mm, blocks);
> +				return err;
> +			}
>   		}
>   next:
>   		gpu_buddy_free_list_internal(mm, blocks);
> @@ -2021,11 +2015,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
>   	min_order = ilog2(min_block_size) - ilog2(mm->chunk_size);
>   
>   	if (order > mm->max_order || size > mm->size) {
> -		if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) &&
> -		    !(flags & GPU_BUDDY_RANGE_ALLOCATION))
> -			return __alloc_contig_try_harder(mm, original_size,
> +		if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
> +			u64 range_start, range_end;
> +
> +			range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? start : 0;
> +			range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : mm->size;
> +
> +			return __alloc_contig_try_harder(mm, range_start,
> +							 range_end,
> +							 original_size,
>   							 original_min_size,
>   							 flags, blocks);
> +		}
>   
>   		return -EINVAL;
>   	}
> @@ -2058,9 +2059,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
>   			 * Try contiguous block allocation through
>   			 * try harder method.
>   			 */
> -			if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION &&
> -			    !(flags & GPU_BUDDY_RANGE_ALLOCATION)) {
> -				err = __alloc_contig_try_harder(mm,
> +			if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
> +				u64 range_start, range_end;
> +
> +				range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? start : 0;
> +				range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : mm->size;
> +
> +				err = __alloc_contig_try_harder(mm, range_start,
> +								range_end,
>   								original_size,
>   								original_min_size,
>   								flags,
> @@ -2068,9 +2074,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
>   				if (!err)
>   					return 0;
>   				if (err != -ENOSPC)
> -					return err;
> -				goto err_free;
> +					goto err_free;
>   			}
> +
>   			err = -ENOSPC;
>   			goto err_free;
>   		} while (1);
> diff --git a/drivers/gpu/tests/gpu_buddy_test.c b/drivers/gpu/tests/gpu_buddy_test.c
> index b75d32ca6ca0..2c445870b808 100644
> --- a/drivers/gpu/tests/gpu_buddy_test.c
> +++ b/drivers/gpu/tests/gpu_buddy_test.c
> @@ -1251,6 +1251,77 @@ static void gpu_test_buddy_alloc_contiguous(struct kunit *test)
>   	gpu_buddy_fini(&mm);
>   }
>   
> +static void gpu_test_buddy_alloc_range_contiguous(struct kunit *test)
> +{
> +	const unsigned long ps = SZ_4K, mm_size = 16 * ps;
> +	const unsigned long range_end = 8 * ps;
> +	struct gpu_buddy_block *block, *prev;
> +	LIST_HEAD(allocated);
> +	struct gpu_buddy mm;
> +	LIST_HEAD(pin_lo);
> +	LIST_HEAD(pin_hi);
> +	u64 total;
> +
> +	KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps),
> +			       "buddy_init failed\n");
> +
> +	/*
> +	 * Idea is to confine the test to the sub-range [0, 32K), which a 12K
> +	 * contiguous request (rounded up to 16K) splits into two naturally
> +	 * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the first 4K page
> +	 * of each slot ([0, 4K) and [16K, 20K)) so that neither aligned slot
> +	 * can satisfy the rounded-up 16K allocation, yet the freed remainder
> +	 * still leaves a contiguous 12K hole at offset 4K, which is page-aligned
> +	 * but not 16K-aligned. A 12K contiguous+range allocation must therefore
> +	 * fall back to stitching that span instead of returning -ENOSPC.
> +	 */
> +	KUNIT_ASSERT_FALSE_MSG(test,
> +			       gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps,
> +						      &pin_lo, 0),
> +			       "failed to pin low page\n");
> +	KUNIT_ASSERT_FALSE_MSG(test,
> +			       gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps,
> +						      ps, &pin_hi, 0),
> +			       "failed to pin high page\n");
> +
> +	/* No aligned 16K block is free; the range-aware fallback must stitch
> +	 * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC.
> +	 */
> +	KUNIT_ASSERT_FALSE_MSG(test,
> +			       gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps,
> +						      ps, &allocated,
> +						      GPU_BUDDY_CONTIGUOUS_ALLOCATION |
> +						      GPU_BUDDY_RANGE_ALLOCATION),
> +			       "range-restricted contiguous alloc failed\n");
> +
> +	/* The result must be exactly 3*ps, contiguous, and inside the range. */
> +	total = 0;
> +	prev = NULL;
> +	list_for_each_entry(block, &allocated, link) {
> +		u64 offset = gpu_buddy_block_offset(block);
> +		u64 bsize = gpu_buddy_block_size(&mm, block);
> +
> +		KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end,
> +				      "block [%llx, %llx) outside range\n",
> +				      offset, offset + bsize);
> +		if (prev)
> +			KUNIT_EXPECT_EQ_MSG(test,
> +					    gpu_buddy_block_offset(prev) +
> +					    gpu_buddy_block_size(&mm, prev),
> +					    offset,
> +					    "block at %llx not contiguous\n",
> +					    offset);
> +		prev = block;
> +		total += bsize;
> +	}
> +	KUNIT_EXPECT_EQ(test, total, 3 * ps);
> +
> +	gpu_buddy_free_list(&mm, &allocated, 0);
> +	gpu_buddy_free_list(&mm, &pin_lo, 0);
> +	gpu_buddy_free_list(&mm, &pin_hi, 0);
> +	gpu_buddy_fini(&mm);
> +}
> +
>   static void gpu_test_buddy_alloc_pathological(struct kunit *test)
>   {
>   	u64 mm_size, size, start = 0;
> @@ -1534,10 +1605,13 @@ static void gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test)
>   				     GPU_BUDDY_RANGE_ALLOCATION);
>   	KUNIT_EXPECT_EQ(test, err, -EINVAL);
>   
> -	/* CONTIGUOUS + RANGE should return -EINVAL (no try_harder for RANGE) */
> -	err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, &blocks,
> -				     GPU_BUDDY_CONTIGUOUS_ALLOCATION | GPU_BUDDY_RANGE_ALLOCATION);
> -	KUNIT_EXPECT_EQ(test, err, -EINVAL);
> +	/* CONTIGUOUS + RANGE should succeed via the range-aware try_harder */
> +	KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, mm_size, size,
> +							    SZ_4K, &blocks,
> +							    GPU_BUDDY_CONTIGUOUS_ALLOCATION |
> +							    GPU_BUDDY_RANGE_ALLOCATION),
> +			       "range contiguous alloc hit an error size=%llu\n", size);
> +	gpu_buddy_free_list(&mm, &blocks, 0);
>   
>   	gpu_buddy_fini(&mm);
>   }
> @@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = {
>   	KUNIT_CASE(gpu_test_buddy_alloc_pessimistic),
>   	KUNIT_CASE(gpu_test_buddy_alloc_pathological),
>   	KUNIT_CASE(gpu_test_buddy_alloc_contiguous),
> +	KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous),
>   	KUNIT_CASE(gpu_test_buddy_alloc_clear),
>   	KUNIT_CASE(gpu_test_buddy_alloc_range),
>   	KUNIT_CASE(gpu_test_buddy_alloc_range_bias),
>
> base-commit: 744f262401cc2e1f3827496c72a47e072d31a852


^ permalink raw reply	[flat|nested] 8+ messages in thread

* Re: [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback
  2026-09-30  6:21 [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback Arunpravin Paneer Selvam
  2026-09-30  6:31 ` sashiko-bot
  2026-10-01  7:06 ` Arunpravin Paneer Selvam
@ 2026-10-01 18:28 ` Matthew Auld
  2026-10-05 14:14   ` Arunpravin Paneer Selvam
  2 siblings, 1 reply; 8+ messages in thread
From: Matthew Auld @ 2026-10-01 18:28 UTC (permalink / raw)
  To: Arunpravin Paneer Selvam, dri-devel, intel-gfx, intel-xe, amd-gfx
  Cc: christian.koenig, alexander.deucher, Anand.Raghavendra

On 30/09/2026 07:21, Arunpravin Paneer Selvam wrote:
> From: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com>
> 
> A range + contiguous allocation (e.g. a scanout FB confined to the
> CPU-visible VRAM aperture) rounds its size up to a power of two and
> requires a naturally aligned free block of that size; on a fragmented
> aperture no such aligned block may exist even though enough contiguous
> space is free, so the allocation fails with -ENOSPC.
> 
> The non-range contiguous path already recovers from this via
> __alloc_contig_try_harder(), which stitches an exact-size span from
> smaller adjacent blocks, but that fallback was unreachable once
> GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder()
> a [range_start, range_end) window and route the range + contiguous
> case through it: each candidate placement is confined to the window
> and aligned to min_block_size. Non-range callers pass [0, mm->size),
> where the guards are no-ops, so the existing behaviour is unchanged.
> A KUnit regression test covers the fragmentation pattern.
> 
> Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b
> regression.
> 
> v2:
>   - Drop the split-undo patch; the range-bias search always descends to
>     an exact-order block or fails the split (already handled), so the
>     extra undo was redundant. (Matthew)
>   - Verified this fallback alone fixes the kms_plane regression.
> 
> Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with decoupled dirty tracker")

Patch looks more like totally new functionally/feature, so the fixes 
here is maybe unexpected? Did something in that fixes commit change 
something such that the try_harder is now needed, but that needs some 
expansion with bias + contig? Do we know exactly what changed here?

> Assisted-by: Claude:claude-opus-4-8

Assisted-by: LLM

> Cc: Matthew Auld <matthew.auld@intel.com>
> Cc: Christian König <christian.koenig@amd.com>
> Signed-off-by: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com>
> ---
>   drivers/gpu/buddy.c                | 88 ++++++++++++++++--------------
>   drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++--
>   2 files changed, 126 insertions(+), 45 deletions(-)
> 
> diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c
> index 2f2aaadafe35..5265e1f6a318 100644
> --- a/drivers/gpu/buddy.c
> +++ b/drivers/gpu/buddy.c
> @@ -1685,26 +1685,14 @@ static int __gpu_buddy_alloc_range(struct gpu_buddy *mm,
>   			     blocks, total_allocated_on_err);
>   }
>   
> -static int __alloc_contig_aligned_retry(struct gpu_buddy *mm,
> -					u64 unaligned_offset,
> -					u64 size,
> -					u64 min_block_size,
> -					unsigned long flags,
> -					struct list_head *blocks)
> -{
> -	u64 aligned_offset = round_down(unaligned_offset, min_block_size);
> -
> -	return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags,
> -				       NULL, blocks);
> -}
> -
>   static int __alloc_contig_try_harder(struct gpu_buddy *mm,
> +				     u64 range_start, u64 range_end,
>   				     u64 size,
>   				     u64 min_block_size,
>   				     unsigned long flags,
>   				     struct list_head *blocks)
>   {
> -	u64 rhs_offset, lhs_offset, filled;
> +	u64 rhs_offset, lhs_offset, filled, aligned;
>   	struct gpu_buddy_block *block;
>   	struct rb_root *root;
>   	struct rb_node *iter;
> @@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct gpu_buddy *mm,
>   					       flags, &filled, blocks);
>   		if (err && err != -ENOSPC)
>   			return err;
> -		if (!err && IS_ALIGNED(rhs_offset, min_block_size))
> +		if (!err && IS_ALIGNED(rhs_offset, min_block_size) &&
> +		    rhs_offset >= range_start && rhs_offset + size <= range_end)
>   			return 0;
>   		if (!err) {
>   			/* Allocate the unaligned RHS offset using round_down */
>   			gpu_buddy_free_list_internal(mm, blocks);
> -			err = __alloc_contig_aligned_retry(mm, rhs_offset,
> -							   size,
> -							   min_block_size,
> -							   flags, blocks);
> -			if (!err)
> -				return 0;
> -			if (err != -ENOSPC) {
> -				gpu_buddy_free_list_internal(mm, blocks);
> -				return err;
> +
> +			aligned = round_down(rhs_offset, min_block_size);
> +			if (aligned >= range_start &&
> +			    aligned + size <= range_end) {
> +				err = __gpu_buddy_alloc_range(mm, aligned, size,
> +							      flags, NULL, blocks);

Did you consider doing a bias-range for [start, end] using 
min_block_size and using what it returns as the starting point, 
extending left/right? If that fails advance start and try again? Maybe 
what you have here is much better for the case you have in mind?

> +				if (!err)
> +					return 0;
> +				if (err != -ENOSPC) {
> +					gpu_buddy_free_list_internal(mm, blocks);
> +					return err;
> +				}
>   			}
>   			goto next;
>   		}
> @@ -1759,15 +1751,17 @@ static int __alloc_contig_try_harder(struct gpu_buddy *mm,
>   
>   		/* Allocate the unaligned LHS offset using round_down */
>   		gpu_buddy_free_list_internal(mm, blocks);
> -		err = __alloc_contig_aligned_retry(mm, lhs_offset,
> -						   size,
> -						   min_block_size,
> -						   flags, blocks);
> -		if (!err)
> -			return 0;
> -		if (err != -ENOSPC) {
> -			gpu_buddy_free_list_internal(mm, blocks);
> -			return err;
> +
> +		aligned = round_down(lhs_offset, min_block_size);
> +		if (aligned >= range_start && aligned + size <= range_end) {
> +			err = __gpu_buddy_alloc_range(mm, aligned, size,
> +						      flags, NULL, blocks);
> +			if (!err)
> +				return 0;
> +			if (err != -ENOSPC) {
> +				gpu_buddy_free_list_internal(mm, blocks);
> +				return err;
> +			}
>   		}
>   next:
>   		gpu_buddy_free_list_internal(mm, blocks);
> @@ -2021,11 +2015,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
>   	min_order = ilog2(min_block_size) - ilog2(mm->chunk_size);
>   
>   	if (order > mm->max_order || size > mm->size) {
> -		if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) &&
> -		    !(flags & GPU_BUDDY_RANGE_ALLOCATION))
> -			return __alloc_contig_try_harder(mm, original_size,
> +		if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
> +			u64 range_start, range_end;
> +
> +			range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? start : 0;
> +			range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : mm->size;
> +
> +			return __alloc_contig_try_harder(mm, range_start,
> +							 range_end,
> +							 original_size,
>   							 original_min_size,
>   							 flags, blocks);
> +		}
>   
>   		return -EINVAL;
>   	}
> @@ -2058,9 +2059,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
>   			 * Try contiguous block allocation through
>   			 * try harder method.
>   			 */
> -			if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION &&
> -			    !(flags & GPU_BUDDY_RANGE_ALLOCATION)) {
> -				err = __alloc_contig_try_harder(mm,
> +			if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
> +				u64 range_start, range_end;
> +
> +				range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? start : 0;
> +				range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : mm->size;
> +
> +				err = __alloc_contig_try_harder(mm, range_start,
> +								range_end,
>   								original_size,
>   								original_min_size,
>   								flags,
> @@ -2068,9 +2074,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
>   				if (!err)
>   					return 0;
>   				if (err != -ENOSPC)
> -					return err;
> -				goto err_free;
> +					goto err_free;
>   			}
> +
>   			err = -ENOSPC;
>   			goto err_free;
>   		} while (1);
> diff --git a/drivers/gpu/tests/gpu_buddy_test.c b/drivers/gpu/tests/gpu_buddy_test.c
> index b75d32ca6ca0..2c445870b808 100644
> --- a/drivers/gpu/tests/gpu_buddy_test.c
> +++ b/drivers/gpu/tests/gpu_buddy_test.c
> @@ -1251,6 +1251,77 @@ static void gpu_test_buddy_alloc_contiguous(struct kunit *test)
>   	gpu_buddy_fini(&mm);
>   }
>   
> +static void gpu_test_buddy_alloc_range_contiguous(struct kunit *test)
> +{
> +	const unsigned long ps = SZ_4K, mm_size = 16 * ps;
> +	const unsigned long range_end = 8 * ps;
> +	struct gpu_buddy_block *block, *prev;
> +	LIST_HEAD(allocated);
> +	struct gpu_buddy mm;
> +	LIST_HEAD(pin_lo);
> +	LIST_HEAD(pin_hi);
> +	u64 total;
> +
> +	KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps),
> +			       "buddy_init failed\n");
> +
> +	/*
> +	 * Idea is to confine the test to the sub-range [0, 32K), which a 12K
> +	 * contiguous request (rounded up to 16K) splits into two naturally
> +	 * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the first 4K page
> +	 * of each slot ([0, 4K) and [16K, 20K)) so that neither aligned slot
> +	 * can satisfy the rounded-up 16K allocation, yet the freed remainder
> +	 * still leaves a contiguous 12K hole at offset 4K, which is page-aligned
> +	 * but not 16K-aligned. A 12K contiguous+range allocation must therefore
> +	 * fall back to stitching that span instead of returning -ENOSPC.
> +	 */
> +	KUNIT_ASSERT_FALSE_MSG(test,
> +			       gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps,
> +						      &pin_lo, 0),
> +			       "failed to pin low page\n");
> +	KUNIT_ASSERT_FALSE_MSG(test,
> +			       gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps,
> +						      ps, &pin_hi, 0),
> +			       "failed to pin high page\n");
> +
> +	/* No aligned 16K block is free; the range-aware fallback must stitch
> +	 * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC.
> +	 */
> +	KUNIT_ASSERT_FALSE_MSG(test,
> +			       gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps,
> +						      ps, &allocated,
> +						      GPU_BUDDY_CONTIGUOUS_ALLOCATION |
> +						      GPU_BUDDY_RANGE_ALLOCATION),
> +			       "range-restricted contiguous alloc failed\n");
> +
> +	/* The result must be exactly 3*ps, contiguous, and inside the range. */
> +	total = 0;
> +	prev = NULL;
> +	list_for_each_entry(block, &allocated, link) {
> +		u64 offset = gpu_buddy_block_offset(block);
> +		u64 bsize = gpu_buddy_block_size(&mm, block);
> +
> +		KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end,
> +				      "block [%llx, %llx) outside range\n",
> +				      offset, offset + bsize);
> +		if (prev)
> +			KUNIT_EXPECT_EQ_MSG(test,
> +					    gpu_buddy_block_offset(prev) +
> +					    gpu_buddy_block_size(&mm, prev),
> +					    offset,
> +					    "block at %llx not contiguous\n",
> +					    offset);
> +		prev = block;
> +		total += bsize;
> +	}
> +	KUNIT_EXPECT_EQ(test, total, 3 * ps);
> +
> +	gpu_buddy_free_list(&mm, &allocated, 0);
> +	gpu_buddy_free_list(&mm, &pin_lo, 0);
> +	gpu_buddy_free_list(&mm, &pin_hi, 0);
> +	gpu_buddy_fini(&mm);
> +}
> +
>   static void gpu_test_buddy_alloc_pathological(struct kunit *test)
>   {
>   	u64 mm_size, size, start = 0;
> @@ -1534,10 +1605,13 @@ static void gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test)
>   				     GPU_BUDDY_RANGE_ALLOCATION);
>   	KUNIT_EXPECT_EQ(test, err, -EINVAL);
>   
> -	/* CONTIGUOUS + RANGE should return -EINVAL (no try_harder for RANGE) */
> -	err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, &blocks,
> -				     GPU_BUDDY_CONTIGUOUS_ALLOCATION | GPU_BUDDY_RANGE_ALLOCATION);
> -	KUNIT_EXPECT_EQ(test, err, -EINVAL);
> +	/* CONTIGUOUS + RANGE should succeed via the range-aware try_harder */
> +	KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, mm_size, size,
> +							    SZ_4K, &blocks,
> +							    GPU_BUDDY_CONTIGUOUS_ALLOCATION |
> +							    GPU_BUDDY_RANGE_ALLOCATION),
> +			       "range contiguous alloc hit an error size=%llu\n", size);
> +	gpu_buddy_free_list(&mm, &blocks, 0);
>   
>   	gpu_buddy_fini(&mm);
>   }
> @@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = {
>   	KUNIT_CASE(gpu_test_buddy_alloc_pessimistic),
>   	KUNIT_CASE(gpu_test_buddy_alloc_pathological),
>   	KUNIT_CASE(gpu_test_buddy_alloc_contiguous),
> +	KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous),
>   	KUNIT_CASE(gpu_test_buddy_alloc_clear),
>   	KUNIT_CASE(gpu_test_buddy_alloc_range),
>   	KUNIT_CASE(gpu_test_buddy_alloc_range_bias),
> 
> base-commit: 744f262401cc2e1f3827496c72a47e072d31a852


^ permalink raw reply	[flat|nested] 8+ messages in thread

* Re: [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback
  2026-10-01 18:28 ` Matthew Auld
@ 2026-10-05 14:14   ` Arunpravin Paneer Selvam
  2026-10-06  9:57     ` Matthew Auld
  0 siblings, 1 reply; 8+ messages in thread
From: Arunpravin Paneer Selvam @ 2026-10-05 14:14 UTC (permalink / raw)
  To: Matthew Auld, dri-devel, intel-gfx, intel-xe, amd-gfx
  Cc: christian.koenig, alexander.deucher, Anand.Raghavendra



On 10/1/2026 11:58 PM, Matthew Auld wrote:
> On 30/09/2026 07:21, Arunpravin Paneer Selvam wrote:
>> From: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com>
>>
>> A range + contiguous allocation (e.g. a scanout FB confined to the
>> CPU-visible VRAM aperture) rounds its size up to a power of two and
>> requires a naturally aligned free block of that size; on a fragmented
>> aperture no such aligned block may exist even though enough contiguous
>> space is free, so the allocation fails with -ENOSPC.
>>
>> The non-range contiguous path already recovers from this via
>> __alloc_contig_try_harder(), which stitches an exact-size span from
>> smaller adjacent blocks, but that fallback was unreachable once
>> GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder()
>> a [range_start, range_end) window and route the range + contiguous
>> case through it: each candidate placement is confined to the window
>> and aligned to min_block_size. Non-range callers pass [0, mm->size),
>> where the guards are no-ops, so the existing behaviour is unchanged.
>> A KUnit regression test covers the fragmentation pattern.
>>
>> Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b
>> regression.
>>
>> v2:
>>   - Drop the split-undo patch; the range-bias search always descends to
>>     an exact-order block or fails the split (already handled), so the
>>     extra undo was redundant. (Matthew)
>>   - Verified this fallback alone fixes the kms_plane regression.
>>
>> Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with 
>> decoupled dirty tracker")
>
> Patch looks more like totally new functionally/feature, so the fixes 
> here is maybe unexpected? Did something in that fixes commit change 
> something such that the try_harder is now needed, but that needs some 
> expansion with bias + contig? Do we know exactly what changed here?
That commit removed __force_merge(), which previously helped recover 
larger contiguous allocations on demand. As a result, some 
range-restricted contiguous allocation requests
that used to succeed now fail with -ENOSPC (the kms_plane regression). 
This patch restores that capability through the try_harder() stitching 
logic, so the Fixes: tag reflects a
regression introduced by that commit rather than new functionality.
>
>> Assisted-by: Claude:claude-opus-4-8
>
> Assisted-by: LLM
>
>> Cc: Matthew Auld <matthew.auld@intel.com>
>> Cc: Christian König <christian.koenig@amd.com>
>> Signed-off-by: Arunpravin Paneer Selvam 
>> <Arunpravin.PaneerSelvam@amd.com>
>> ---
>>   drivers/gpu/buddy.c                | 88 ++++++++++++++++--------------
>>   drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++--
>>   2 files changed, 126 insertions(+), 45 deletions(-)
>>
>> diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c
>> index 2f2aaadafe35..5265e1f6a318 100644
>> --- a/drivers/gpu/buddy.c
>> +++ b/drivers/gpu/buddy.c
>> @@ -1685,26 +1685,14 @@ static int __gpu_buddy_alloc_range(struct 
>> gpu_buddy *mm,
>>                    blocks, total_allocated_on_err);
>>   }
>>   -static int __alloc_contig_aligned_retry(struct gpu_buddy *mm,
>> -                    u64 unaligned_offset,
>> -                    u64 size,
>> -                    u64 min_block_size,
>> -                    unsigned long flags,
>> -                    struct list_head *blocks)
>> -{
>> -    u64 aligned_offset = round_down(unaligned_offset, min_block_size);
>> -
>> -    return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags,
>> -                       NULL, blocks);
>> -}
>> -
>>   static int __alloc_contig_try_harder(struct gpu_buddy *mm,
>> +                     u64 range_start, u64 range_end,
>>                        u64 size,
>>                        u64 min_block_size,
>>                        unsigned long flags,
>>                        struct list_head *blocks)
>>   {
>> -    u64 rhs_offset, lhs_offset, filled;
>> +    u64 rhs_offset, lhs_offset, filled, aligned;
>>       struct gpu_buddy_block *block;
>>       struct rb_root *root;
>>       struct rb_node *iter;
>> @@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct 
>> gpu_buddy *mm,
>>                              flags, &filled, blocks);
>>           if (err && err != -ENOSPC)
>>               return err;
>> -        if (!err && IS_ALIGNED(rhs_offset, min_block_size))
>> +        if (!err && IS_ALIGNED(rhs_offset, min_block_size) &&
>> +            rhs_offset >= range_start && rhs_offset + size <= 
>> range_end)
>>               return 0;
>>           if (!err) {
>>               /* Allocate the unaligned RHS offset using round_down */
>>               gpu_buddy_free_list_internal(mm, blocks);
>> -            err = __alloc_contig_aligned_retry(mm, rhs_offset,
>> -                               size,
>> -                               min_block_size,
>> -                               flags, blocks);
>> -            if (!err)
>> -                return 0;
>> -            if (err != -ENOSPC) {
>> -                gpu_buddy_free_list_internal(mm, blocks);
>> -                return err;
>> +
>> +            aligned = round_down(rhs_offset, min_block_size);
>> +            if (aligned >= range_start &&
>> +                aligned + size <= range_end) {
>> +                err = __gpu_buddy_alloc_range(mm, aligned, size,
>> +                                  flags, NULL, blocks);
>
> Did you consider doing a bias-range for [start, end] using 
> min_block_size and using what it returns as the starting point, 
> extending left/right? If that fails advance start and try again? Maybe 
> what you have here is much better for the case you have in mind?
I took a closer look at the alloc_range_bias approach. My understanding 
is that this would repeatedly bias within [start, end], use the returned 
min_block_size block as a starting point, and then extend left/right to 
build the requested run.
While that should work, every failed attempt may require splitting 
higher-order blocks to obtain a min_block_size block, followed by a 
free/merge back when the extension fails.
The free-tree descent evaluates the available starting points through a 
tree walk without any splitting or allocation just for enumeration, so I 
kept that approach.
Please let me know if there is a case where alloc_range_bias would 
provide an advantage over the free-tree descent.

While evaluating it, I found two issues in the current try_harder() 
implementation:

1. It only searches free_tree[size_order]. This is insufficient when 
min_block_size < size. For example, for a 128 KiB allocation (order-5) 
with min_block_size = 4 KiB, free_tree[5] may be empty
while two adjacent order-4 (64 KiB) blocks covering [64 KiB, 128 KiB] 
and [128 KiB, 192 KiB] are free. These blocks still form a valid 
order-5-sized contiguous run, but they can never appear in
free_tree[5] because they are not mergeable buddies. As a result, a 
search that only considers free_tree[5] will never find this placement. 
v3 addresses this by walking the free tree from the
requested size order down to min_order, while reusing the existing 
extend/stitch logic and min_block_size alignment checks unchanged.

2. The search is not restricted to the requested range. v3 confines the 
walk to [range_start, range_end], skipping candidates at or above 
range_end and stopping once the walk falls below range_start.
This ensures that range-constrained allocations only consider free 
blocks within the requested window and avoids performing allocation/free 
attempts on blocks outside the specified range.
For non-range callers, the effective window remains [0, mm->size], so 
the behavior is unchanged.

Regards,
Arun.
>
>> +                if (!err)
>> +                    return 0;
>> +                if (err != -ENOSPC) {
>> +                    gpu_buddy_free_list_internal(mm, blocks);
>> +                    return err;
>> +                }
>>               }
>>               goto next;
>>           }
>> @@ -1759,15 +1751,17 @@ static int __alloc_contig_try_harder(struct 
>> gpu_buddy *mm,
>>             /* Allocate the unaligned LHS offset using round_down */
>>           gpu_buddy_free_list_internal(mm, blocks);
>> -        err = __alloc_contig_aligned_retry(mm, lhs_offset,
>> -                           size,
>> -                           min_block_size,
>> -                           flags, blocks);
>> -        if (!err)
>> -            return 0;
>> -        if (err != -ENOSPC) {
>> -            gpu_buddy_free_list_internal(mm, blocks);
>> -            return err;
>> +
>> +        aligned = round_down(lhs_offset, min_block_size);
>> +        if (aligned >= range_start && aligned + size <= range_end) {
>> +            err = __gpu_buddy_alloc_range(mm, aligned, size,
>> +                              flags, NULL, blocks);
>> +            if (!err)
>> +                return 0;
>> +            if (err != -ENOSPC) {
>> +                gpu_buddy_free_list_internal(mm, blocks);
>> +                return err;
>> +            }
>>           }
>>   next:
>>           gpu_buddy_free_list_internal(mm, blocks);
>> @@ -2021,11 +2015,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
>>       min_order = ilog2(min_block_size) - ilog2(mm->chunk_size);
>>         if (order > mm->max_order || size > mm->size) {
>> -        if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) &&
>> -            !(flags & GPU_BUDDY_RANGE_ALLOCATION))
>> -            return __alloc_contig_try_harder(mm, original_size,
>> +        if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
>> +            u64 range_start, range_end;
>> +
>> +            range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? 
>> start : 0;
>> +            range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : 
>> mm->size;
>> +
>> +            return __alloc_contig_try_harder(mm, range_start,
>> +                             range_end,
>> +                             original_size,
>>                                original_min_size,
>>                                flags, blocks);
>> +        }
>>             return -EINVAL;
>>       }
>> @@ -2058,9 +2059,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
>>                * Try contiguous block allocation through
>>                * try harder method.
>>                */
>> -            if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION &&
>> -                !(flags & GPU_BUDDY_RANGE_ALLOCATION)) {
>> -                err = __alloc_contig_try_harder(mm,
>> +            if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
>> +                u64 range_start, range_end;
>> +
>> +                range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? 
>> start : 0;
>> +                range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? 
>> end : mm->size;
>> +
>> +                err = __alloc_contig_try_harder(mm, range_start,
>> +                                range_end,
>>                                   original_size,
>>                                   original_min_size,
>>                                   flags,
>> @@ -2068,9 +2074,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
>>                   if (!err)
>>                       return 0;
>>                   if (err != -ENOSPC)
>> -                    return err;
>> -                goto err_free;
>> +                    goto err_free;
>>               }
>> +
>>               err = -ENOSPC;
>>               goto err_free;
>>           } while (1);
>> diff --git a/drivers/gpu/tests/gpu_buddy_test.c 
>> b/drivers/gpu/tests/gpu_buddy_test.c
>> index b75d32ca6ca0..2c445870b808 100644
>> --- a/drivers/gpu/tests/gpu_buddy_test.c
>> +++ b/drivers/gpu/tests/gpu_buddy_test.c
>> @@ -1251,6 +1251,77 @@ static void 
>> gpu_test_buddy_alloc_contiguous(struct kunit *test)
>>       gpu_buddy_fini(&mm);
>>   }
>>   +static void gpu_test_buddy_alloc_range_contiguous(struct kunit *test)
>> +{
>> +    const unsigned long ps = SZ_4K, mm_size = 16 * ps;
>> +    const unsigned long range_end = 8 * ps;
>> +    struct gpu_buddy_block *block, *prev;
>> +    LIST_HEAD(allocated);
>> +    struct gpu_buddy mm;
>> +    LIST_HEAD(pin_lo);
>> +    LIST_HEAD(pin_hi);
>> +    u64 total;
>> +
>> +    KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps),
>> +                   "buddy_init failed\n");
>> +
>> +    /*
>> +     * Idea is to confine the test to the sub-range [0, 32K), which 
>> a 12K
>> +     * contiguous request (rounded up to 16K) splits into two naturally
>> +     * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the first 
>> 4K page
>> +     * of each slot ([0, 4K) and [16K, 20K)) so that neither aligned 
>> slot
>> +     * can satisfy the rounded-up 16K allocation, yet the freed 
>> remainder
>> +     * still leaves a contiguous 12K hole at offset 4K, which is 
>> page-aligned
>> +     * but not 16K-aligned. A 12K contiguous+range allocation must 
>> therefore
>> +     * fall back to stitching that span instead of returning -ENOSPC.
>> +     */
>> +    KUNIT_ASSERT_FALSE_MSG(test,
>> +                   gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps,
>> +                              &pin_lo, 0),
>> +                   "failed to pin low page\n");
>> +    KUNIT_ASSERT_FALSE_MSG(test,
>> +                   gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps,
>> +                              ps, &pin_hi, 0),
>> +                   "failed to pin high page\n");
>> +
>> +    /* No aligned 16K block is free; the range-aware fallback must 
>> stitch
>> +     * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC.
>> +     */
>> +    KUNIT_ASSERT_FALSE_MSG(test,
>> +                   gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps,
>> +                              ps, &allocated,
>> +                              GPU_BUDDY_CONTIGUOUS_ALLOCATION |
>> +                              GPU_BUDDY_RANGE_ALLOCATION),
>> +                   "range-restricted contiguous alloc failed\n");
>> +
>> +    /* The result must be exactly 3*ps, contiguous, and inside the 
>> range. */
>> +    total = 0;
>> +    prev = NULL;
>> +    list_for_each_entry(block, &allocated, link) {
>> +        u64 offset = gpu_buddy_block_offset(block);
>> +        u64 bsize = gpu_buddy_block_size(&mm, block);
>> +
>> +        KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end,
>> +                      "block [%llx, %llx) outside range\n",
>> +                      offset, offset + bsize);
>> +        if (prev)
>> +            KUNIT_EXPECT_EQ_MSG(test,
>> +                        gpu_buddy_block_offset(prev) +
>> +                        gpu_buddy_block_size(&mm, prev),
>> +                        offset,
>> +                        "block at %llx not contiguous\n",
>> +                        offset);
>> +        prev = block;
>> +        total += bsize;
>> +    }
>> +    KUNIT_EXPECT_EQ(test, total, 3 * ps);
>> +
>> +    gpu_buddy_free_list(&mm, &allocated, 0);
>> +    gpu_buddy_free_list(&mm, &pin_lo, 0);
>> +    gpu_buddy_free_list(&mm, &pin_hi, 0);
>> +    gpu_buddy_fini(&mm);
>> +}
>> +
>>   static void gpu_test_buddy_alloc_pathological(struct kunit *test)
>>   {
>>       u64 mm_size, size, start = 0;
>> @@ -1534,10 +1605,13 @@ static void 
>> gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test)
>>                        GPU_BUDDY_RANGE_ALLOCATION);
>>       KUNIT_EXPECT_EQ(test, err, -EINVAL);
>>   -    /* CONTIGUOUS + RANGE should return -EINVAL (no try_harder for 
>> RANGE) */
>> -    err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, &blocks,
>> -                     GPU_BUDDY_CONTIGUOUS_ALLOCATION | 
>> GPU_BUDDY_RANGE_ALLOCATION);
>> -    KUNIT_EXPECT_EQ(test, err, -EINVAL);
>> +    /* CONTIGUOUS + RANGE should succeed via the range-aware 
>> try_harder */
>> +    KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, 
>> mm_size, size,
>> +                                SZ_4K, &blocks,
>> +                                GPU_BUDDY_CONTIGUOUS_ALLOCATION |
>> +                                GPU_BUDDY_RANGE_ALLOCATION),
>> +                   "range contiguous alloc hit an error 
>> size=%llu\n", size);
>> +    gpu_buddy_free_list(&mm, &blocks, 0);
>>         gpu_buddy_fini(&mm);
>>   }
>> @@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = {
>>       KUNIT_CASE(gpu_test_buddy_alloc_pessimistic),
>>       KUNIT_CASE(gpu_test_buddy_alloc_pathological),
>>       KUNIT_CASE(gpu_test_buddy_alloc_contiguous),
>> +    KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous),
>>       KUNIT_CASE(gpu_test_buddy_alloc_clear),
>>       KUNIT_CASE(gpu_test_buddy_alloc_range),
>>       KUNIT_CASE(gpu_test_buddy_alloc_range_bias),
>>
>> base-commit: 744f262401cc2e1f3827496c72a47e072d31a852
>


^ permalink raw reply	[flat|nested] 8+ messages in thread

* Re: [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback
  2026-10-05 14:14   ` Arunpravin Paneer Selvam
@ 2026-10-06  9:57     ` Matthew Auld
  2026-10-09  9:14       ` Arunpravin Paneer Selvam
  0 siblings, 1 reply; 8+ messages in thread
From: Matthew Auld @ 2026-10-06  9:57 UTC (permalink / raw)
  To: Arunpravin Paneer Selvam, dri-devel, intel-gfx, intel-xe, amd-gfx
  Cc: christian.koenig, alexander.deucher, Anand.Raghavendra

On 05/10/2026 15:14, Arunpravin Paneer Selvam wrote:
> 
> 
> On 10/1/2026 11:58 PM, Matthew Auld wrote:
>> On 30/09/2026 07:21, Arunpravin Paneer Selvam wrote:
>>> From: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com>
>>>
>>> A range + contiguous allocation (e.g. a scanout FB confined to the
>>> CPU-visible VRAM aperture) rounds its size up to a power of two and
>>> requires a naturally aligned free block of that size; on a fragmented
>>> aperture no such aligned block may exist even though enough contiguous
>>> space is free, so the allocation fails with -ENOSPC.
>>>
>>> The non-range contiguous path already recovers from this via
>>> __alloc_contig_try_harder(), which stitches an exact-size span from
>>> smaller adjacent blocks, but that fallback was unreachable once
>>> GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder()
>>> a [range_start, range_end) window and route the range + contiguous
>>> case through it: each candidate placement is confined to the window
>>> and aligned to min_block_size. Non-range callers pass [0, mm->size),
>>> where the guards are no-ops, so the existing behaviour is unchanged.
>>> A KUnit regression test covers the fragmentation pattern.
>>>
>>> Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b
>>> regression.
>>>
>>> v2:
>>>   - Drop the split-undo patch; the range-bias search always descends to
>>>     an exact-order block or fails the split (already handled), so the
>>>     extra undo was redundant. (Matthew)
>>>   - Verified this fallback alone fixes the kms_plane regression.
>>>
>>> Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with 
>>> decoupled dirty tracker")
>>
>> Patch looks more like totally new functionally/feature, so the fixes 
>> here is maybe unexpected? Did something in that fixes commit change 
>> something such that the try_harder is now needed, but that needs some 
>> expansion with bias + contig? Do we know exactly what changed here?
> That commit removed __force_merge(), which previously helped recover 
> larger contiguous allocations on demand. As a result, some range- 

__force_merge() got nuked, but IIRC I think that was essentially because 
we now "force merge" on free, so shouldn't we get the ~same result? Or 
is the fact that we force_merge() on every free giving a different 
layout of pages, and with some very some specific allocation pattern 
that difference subtly results in -ENOSPC, somehow?

> restricted contiguous allocation requests
> that used to succeed now fail with -ENOSPC (the kms_plane regression). 
> This patch restores that capability through the try_harder() stitching 
> logic, so the Fixes: tag reflects a
> regression introduced by that commit rather than new functionality.
>>
>>> Assisted-by: Claude:claude-opus-4-8
>>
>> Assisted-by: LLM
>>
>>> Cc: Matthew Auld <matthew.auld@intel.com>
>>> Cc: Christian König <christian.koenig@amd.com>
>>> Signed-off-by: Arunpravin Paneer Selvam 
>>> <Arunpravin.PaneerSelvam@amd.com>
>>> ---
>>>   drivers/gpu/buddy.c                | 88 ++++++++++++++++--------------
>>>   drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++--
>>>   2 files changed, 126 insertions(+), 45 deletions(-)
>>>
>>> diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c
>>> index 2f2aaadafe35..5265e1f6a318 100644
>>> --- a/drivers/gpu/buddy.c
>>> +++ b/drivers/gpu/buddy.c
>>> @@ -1685,26 +1685,14 @@ static int __gpu_buddy_alloc_range(struct 
>>> gpu_buddy *mm,
>>>                    blocks, total_allocated_on_err);
>>>   }
>>>   -static int __alloc_contig_aligned_retry(struct gpu_buddy *mm,
>>> -                    u64 unaligned_offset,
>>> -                    u64 size,
>>> -                    u64 min_block_size,
>>> -                    unsigned long flags,
>>> -                    struct list_head *blocks)
>>> -{
>>> -    u64 aligned_offset = round_down(unaligned_offset, min_block_size);
>>> -
>>> -    return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags,
>>> -                       NULL, blocks);
>>> -}
>>> -
>>>   static int __alloc_contig_try_harder(struct gpu_buddy *mm,
>>> +                     u64 range_start, u64 range_end,
>>>                        u64 size,
>>>                        u64 min_block_size,
>>>                        unsigned long flags,
>>>                        struct list_head *blocks)
>>>   {
>>> -    u64 rhs_offset, lhs_offset, filled;
>>> +    u64 rhs_offset, lhs_offset, filled, aligned;
>>>       struct gpu_buddy_block *block;
>>>       struct rb_root *root;
>>>       struct rb_node *iter;
>>> @@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct 
>>> gpu_buddy *mm,
>>>                              flags, &filled, blocks);
>>>           if (err && err != -ENOSPC)
>>>               return err;
>>> -        if (!err && IS_ALIGNED(rhs_offset, min_block_size))
>>> +        if (!err && IS_ALIGNED(rhs_offset, min_block_size) &&
>>> +            rhs_offset >= range_start && rhs_offset + size <= 
>>> range_end)
>>>               return 0;
>>>           if (!err) {
>>>               /* Allocate the unaligned RHS offset using round_down */
>>>               gpu_buddy_free_list_internal(mm, blocks);
>>> -            err = __alloc_contig_aligned_retry(mm, rhs_offset,
>>> -                               size,
>>> -                               min_block_size,
>>> -                               flags, blocks);
>>> -            if (!err)
>>> -                return 0;
>>> -            if (err != -ENOSPC) {
>>> -                gpu_buddy_free_list_internal(mm, blocks);
>>> -                return err;
>>> +
>>> +            aligned = round_down(rhs_offset, min_block_size);
>>> +            if (aligned >= range_start &&
>>> +                aligned + size <= range_end) {
>>> +                err = __gpu_buddy_alloc_range(mm, aligned, size,
>>> +                                  flags, NULL, blocks);
>>
>> Did you consider doing a bias-range for [start, end] using 
>> min_block_size and using what it returns as the starting point, 
>> extending left/right? If that fails advance start and try again? Maybe 
>> what you have here is much better for the case you have in mind?
> I took a closer look at the alloc_range_bias approach. My understanding 
> is that this would repeatedly bias within [start, end], use the returned 
> min_block_size block as a starting point, and then extend left/right to 
> build the requested run.
> While that should work, every failed attempt may require splitting 
> higher-order blocks to obtain a min_block_size block, followed by a 
> free/merge back when the extension fails.
> The free-tree descent evaluates the available starting points through a 
> tree walk without any splitting or allocation just for enumeration, so I 
> kept that approach.
> Please let me know if there is a case where alloc_range_bias would 
> provide an advantage over the free-tree descent.
> 
> While evaluating it, I found two issues in the current try_harder() 
> implementation:
> 
> 1. It only searches free_tree[size_order]. This is insufficient when 
> min_block_size < size. For example, for a 128 KiB allocation (order-5) 
> with min_block_size = 4 KiB, free_tree[5] may be empty
> while two adjacent order-4 (64 KiB) blocks covering [64 KiB, 128 KiB] 
> and [128 KiB, 192 KiB] are free. These blocks still form a valid 
> order-5-sized contiguous run, but they can never appear in
> free_tree[5] because they are not mergeable buddies. As a result, a 
> search that only considers free_tree[5] will never find this placement. 
> v3 addresses this by walking the free tree from the
> requested size order down to min_order, while reusing the existing 
> extend/stitch logic and min_block_size alignment checks unchanged.
> 
> 2. The search is not restricted to the requested range. v3 confines the 
> walk to [range_start, range_end], skipping candidates at or above 
> range_end and stopping once the walk falls below range_start.
> This ensures that range-constrained allocations only consider free 
> blocks within the requested window and avoids performing allocation/free 
> attempts on blocks outside the specified range.
> For non-range callers, the effective window remains [0, mm->size], so 
> the behavior is unchanged.
> 
> Regards,
> Arun.
>>
>>> +                if (!err)
>>> +                    return 0;
>>> +                if (err != -ENOSPC) {
>>> +                    gpu_buddy_free_list_internal(mm, blocks);
>>> +                    return err;
>>> +                }
>>>               }
>>>               goto next;
>>>           }
>>> @@ -1759,15 +1751,17 @@ static int __alloc_contig_try_harder(struct 
>>> gpu_buddy *mm,
>>>             /* Allocate the unaligned LHS offset using round_down */
>>>           gpu_buddy_free_list_internal(mm, blocks);
>>> -        err = __alloc_contig_aligned_retry(mm, lhs_offset,
>>> -                           size,
>>> -                           min_block_size,
>>> -                           flags, blocks);
>>> -        if (!err)
>>> -            return 0;
>>> -        if (err != -ENOSPC) {
>>> -            gpu_buddy_free_list_internal(mm, blocks);
>>> -            return err;
>>> +
>>> +        aligned = round_down(lhs_offset, min_block_size);
>>> +        if (aligned >= range_start && aligned + size <= range_end) {
>>> +            err = __gpu_buddy_alloc_range(mm, aligned, size,
>>> +                              flags, NULL, blocks);
>>> +            if (!err)
>>> +                return 0;
>>> +            if (err != -ENOSPC) {
>>> +                gpu_buddy_free_list_internal(mm, blocks);
>>> +                return err;
>>> +            }
>>>           }
>>>   next:
>>>           gpu_buddy_free_list_internal(mm, blocks);
>>> @@ -2021,11 +2015,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
>>>       min_order = ilog2(min_block_size) - ilog2(mm->chunk_size);
>>>         if (order > mm->max_order || size > mm->size) {
>>> -        if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) &&
>>> -            !(flags & GPU_BUDDY_RANGE_ALLOCATION))
>>> -            return __alloc_contig_try_harder(mm, original_size,
>>> +        if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
>>> +            u64 range_start, range_end;
>>> +
>>> +            range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? 
>>> start : 0;
>>> +            range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : 
>>> mm->size;
>>> +
>>> +            return __alloc_contig_try_harder(mm, range_start,
>>> +                             range_end,
>>> +                             original_size,
>>>                                original_min_size,
>>>                                flags, blocks);
>>> +        }
>>>             return -EINVAL;
>>>       }
>>> @@ -2058,9 +2059,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
>>>                * Try contiguous block allocation through
>>>                * try harder method.
>>>                */
>>> -            if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION &&
>>> -                !(flags & GPU_BUDDY_RANGE_ALLOCATION)) {
>>> -                err = __alloc_contig_try_harder(mm,
>>> +            if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
>>> +                u64 range_start, range_end;
>>> +
>>> +                range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? 
>>> start : 0;
>>> +                range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? 
>>> end : mm->size;
>>> +
>>> +                err = __alloc_contig_try_harder(mm, range_start,
>>> +                                range_end,
>>>                                   original_size,
>>>                                   original_min_size,
>>>                                   flags,
>>> @@ -2068,9 +2074,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
>>>                   if (!err)
>>>                       return 0;
>>>                   if (err != -ENOSPC)
>>> -                    return err;
>>> -                goto err_free;
>>> +                    goto err_free;
>>>               }
>>> +
>>>               err = -ENOSPC;
>>>               goto err_free;
>>>           } while (1);
>>> diff --git a/drivers/gpu/tests/gpu_buddy_test.c b/drivers/gpu/tests/ 
>>> gpu_buddy_test.c
>>> index b75d32ca6ca0..2c445870b808 100644
>>> --- a/drivers/gpu/tests/gpu_buddy_test.c
>>> +++ b/drivers/gpu/tests/gpu_buddy_test.c
>>> @@ -1251,6 +1251,77 @@ static void 
>>> gpu_test_buddy_alloc_contiguous(struct kunit *test)
>>>       gpu_buddy_fini(&mm);
>>>   }
>>>   +static void gpu_test_buddy_alloc_range_contiguous(struct kunit *test)
>>> +{
>>> +    const unsigned long ps = SZ_4K, mm_size = 16 * ps;
>>> +    const unsigned long range_end = 8 * ps;
>>> +    struct gpu_buddy_block *block, *prev;
>>> +    LIST_HEAD(allocated);
>>> +    struct gpu_buddy mm;
>>> +    LIST_HEAD(pin_lo);
>>> +    LIST_HEAD(pin_hi);
>>> +    u64 total;
>>> +
>>> +    KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps),
>>> +                   "buddy_init failed\n");
>>> +
>>> +    /*
>>> +     * Idea is to confine the test to the sub-range [0, 32K), which 
>>> a 12K
>>> +     * contiguous request (rounded up to 16K) splits into two naturally
>>> +     * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the first 
>>> 4K page
>>> +     * of each slot ([0, 4K) and [16K, 20K)) so that neither aligned 
>>> slot
>>> +     * can satisfy the rounded-up 16K allocation, yet the freed 
>>> remainder
>>> +     * still leaves a contiguous 12K hole at offset 4K, which is 
>>> page-aligned
>>> +     * but not 16K-aligned. A 12K contiguous+range allocation must 
>>> therefore
>>> +     * fall back to stitching that span instead of returning -ENOSPC.
>>> +     */
>>> +    KUNIT_ASSERT_FALSE_MSG(test,
>>> +                   gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps,
>>> +                              &pin_lo, 0),
>>> +                   "failed to pin low page\n");
>>> +    KUNIT_ASSERT_FALSE_MSG(test,
>>> +                   gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps,
>>> +                              ps, &pin_hi, 0),
>>> +                   "failed to pin high page\n");
>>> +
>>> +    /* No aligned 16K block is free; the range-aware fallback must 
>>> stitch
>>> +     * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC.
>>> +     */
>>> +    KUNIT_ASSERT_FALSE_MSG(test,
>>> +                   gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps,
>>> +                              ps, &allocated,
>>> +                              GPU_BUDDY_CONTIGUOUS_ALLOCATION |
>>> +                              GPU_BUDDY_RANGE_ALLOCATION),
>>> +                   "range-restricted contiguous alloc failed\n");
>>> +
>>> +    /* The result must be exactly 3*ps, contiguous, and inside the 
>>> range. */
>>> +    total = 0;
>>> +    prev = NULL;
>>> +    list_for_each_entry(block, &allocated, link) {
>>> +        u64 offset = gpu_buddy_block_offset(block);
>>> +        u64 bsize = gpu_buddy_block_size(&mm, block);
>>> +
>>> +        KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end,
>>> +                      "block [%llx, %llx) outside range\n",
>>> +                      offset, offset + bsize);
>>> +        if (prev)
>>> +            KUNIT_EXPECT_EQ_MSG(test,
>>> +                        gpu_buddy_block_offset(prev) +
>>> +                        gpu_buddy_block_size(&mm, prev),
>>> +                        offset,
>>> +                        "block at %llx not contiguous\n",
>>> +                        offset);
>>> +        prev = block;
>>> +        total += bsize;
>>> +    }
>>> +    KUNIT_EXPECT_EQ(test, total, 3 * ps);
>>> +
>>> +    gpu_buddy_free_list(&mm, &allocated, 0);
>>> +    gpu_buddy_free_list(&mm, &pin_lo, 0);
>>> +    gpu_buddy_free_list(&mm, &pin_hi, 0);
>>> +    gpu_buddy_fini(&mm);
>>> +}
>>> +
>>>   static void gpu_test_buddy_alloc_pathological(struct kunit *test)
>>>   {
>>>       u64 mm_size, size, start = 0;
>>> @@ -1534,10 +1605,13 @@ static void 
>>> gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test)
>>>                        GPU_BUDDY_RANGE_ALLOCATION);
>>>       KUNIT_EXPECT_EQ(test, err, -EINVAL);
>>>   -    /* CONTIGUOUS + RANGE should return -EINVAL (no try_harder for 
>>> RANGE) */
>>> -    err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, &blocks,
>>> -                     GPU_BUDDY_CONTIGUOUS_ALLOCATION | 
>>> GPU_BUDDY_RANGE_ALLOCATION);
>>> -    KUNIT_EXPECT_EQ(test, err, -EINVAL);
>>> +    /* CONTIGUOUS + RANGE should succeed via the range-aware 
>>> try_harder */
>>> +    KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, 
>>> mm_size, size,
>>> +                                SZ_4K, &blocks,
>>> +                                GPU_BUDDY_CONTIGUOUS_ALLOCATION |
>>> +                                GPU_BUDDY_RANGE_ALLOCATION),
>>> +                   "range contiguous alloc hit an error 
>>> size=%llu\n", size);
>>> +    gpu_buddy_free_list(&mm, &blocks, 0);
>>>         gpu_buddy_fini(&mm);
>>>   }
>>> @@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = {
>>>       KUNIT_CASE(gpu_test_buddy_alloc_pessimistic),
>>>       KUNIT_CASE(gpu_test_buddy_alloc_pathological),
>>>       KUNIT_CASE(gpu_test_buddy_alloc_contiguous),
>>> +    KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous),
>>>       KUNIT_CASE(gpu_test_buddy_alloc_clear),
>>>       KUNIT_CASE(gpu_test_buddy_alloc_range),
>>>       KUNIT_CASE(gpu_test_buddy_alloc_range_bias),
>>>
>>> base-commit: 744f262401cc2e1f3827496c72a47e072d31a852
>>
> 


^ permalink raw reply	[flat|nested] 8+ messages in thread

* Re: [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback
  2026-10-06  9:57     ` Matthew Auld
@ 2026-10-09  9:14       ` Arunpravin Paneer Selvam
  2026-10-09 10:17         ` Matthew Auld
  0 siblings, 1 reply; 8+ messages in thread
From: Arunpravin Paneer Selvam @ 2026-10-09  9:14 UTC (permalink / raw)
  To: Matthew Auld, dri-devel, intel-gfx, intel-xe, amd-gfx
  Cc: christian.koenig, alexander.deucher, Anand.Raghavendra



On 10/6/2026 3:27 PM, Matthew Auld wrote:
> On 05/10/2026 15:14, Arunpravin Paneer Selvam wrote:
>>
>>
>> On 10/1/2026 11:58 PM, Matthew Auld wrote:
>>> On 30/09/2026 07:21, Arunpravin Paneer Selvam wrote:
>>>> From: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com>
>>>>
>>>> A range + contiguous allocation (e.g. a scanout FB confined to the
>>>> CPU-visible VRAM aperture) rounds its size up to a power of two and
>>>> requires a naturally aligned free block of that size; on a fragmented
>>>> aperture no such aligned block may exist even though enough contiguous
>>>> space is free, so the allocation fails with -ENOSPC.
>>>>
>>>> The non-range contiguous path already recovers from this via
>>>> __alloc_contig_try_harder(), which stitches an exact-size span from
>>>> smaller adjacent blocks, but that fallback was unreachable once
>>>> GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder()
>>>> a [range_start, range_end) window and route the range + contiguous
>>>> case through it: each candidate placement is confined to the window
>>>> and aligned to min_block_size. Non-range callers pass [0, mm->size),
>>>> where the guards are no-ops, so the existing behaviour is unchanged.
>>>> A KUnit regression test covers the fragmentation pattern.
>>>>
>>>> Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b
>>>> regression.
>>>>
>>>> v2:
>>>>   - Drop the split-undo patch; the range-bias search always 
>>>> descends to
>>>>     an exact-order block or fails the split (already handled), so the
>>>>     extra undo was redundant. (Matthew)
>>>>   - Verified this fallback alone fixes the kms_plane regression.
>>>>
>>>> Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with 
>>>> decoupled dirty tracker")
>>>
>>> Patch looks more like totally new functionally/feature, so the fixes 
>>> here is maybe unexpected? Did something in that fixes commit change 
>>> something such that the try_harder is now needed, but that needs 
>>> some expansion with bias + contig? Do we know exactly what changed 
>>> here?
>> That commit removed __force_merge(), which previously helped recover 
>> larger contiguous allocations on demand. As a result, some range- 
>
> __force_merge() got nuked, but IIRC I think that was essentially 
> because we now "force merge" on free, so shouldn't we get the ~same 
> result? Or is the fact that we force_merge() on every free giving a 
> different layout of pages, and with some very some specific allocation 
> pattern that difference subtly results in -ENOSPC, somehow?
You are right - merge-on-free reclaims contiguity just like the old 
on-demand  __force_merge() , so that is not the cause and the layout is 
effectively the same. The real change is where clear/dirty segregation 
lives: it used to be an inline guard inside __alloc_range_bias() , but 
the rework pulled it out into dirty_steer_window() , which only runs for 
non-range allocations. So range (aperture) requests now go through  
__alloc_range_bias()  with no clear/dirty guard at all - clear and dirty 
allocations interleave and fragment the aperture until no large 
same-class run survives for a 4K scanout pin (-> -ENOSPC). Restoring 
that two-pass guard is the real fix.

I have accordingly moved the  Fixes:  tag onto that guard-restore patch, 
and kept the range support in  __alloc_contig_try_harder() as a new 
feature -  it still helps, but it is a safety-net fallback that should 
land on top, not as the root-cause fix.

Regards,
Arun.
>
>> restricted contiguous allocation requests
>> that used to succeed now fail with -ENOSPC (the kms_plane 
>> regression). This patch restores that capability through the 
>> try_harder() stitching logic, so the Fixes: tag reflects a
>> regression introduced by that commit rather than new functionality.
>>>
>>>> Assisted-by: Claude:claude-opus-4-8
>>>
>>> Assisted-by: LLM
>>>
>>>> Cc: Matthew Auld <matthew.auld@intel.com>
>>>> Cc: Christian König <christian.koenig@amd.com>
>>>> Signed-off-by: Arunpravin Paneer Selvam 
>>>> <Arunpravin.PaneerSelvam@amd.com>
>>>> ---
>>>>   drivers/gpu/buddy.c                | 88 
>>>> ++++++++++++++++--------------
>>>>   drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++--
>>>>   2 files changed, 126 insertions(+), 45 deletions(-)
>>>>
>>>> diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c
>>>> index 2f2aaadafe35..5265e1f6a318 100644
>>>> --- a/drivers/gpu/buddy.c
>>>> +++ b/drivers/gpu/buddy.c
>>>> @@ -1685,26 +1685,14 @@ static int __gpu_buddy_alloc_range(struct 
>>>> gpu_buddy *mm,
>>>>                    blocks, total_allocated_on_err);
>>>>   }
>>>>   -static int __alloc_contig_aligned_retry(struct gpu_buddy *mm,
>>>> -                    u64 unaligned_offset,
>>>> -                    u64 size,
>>>> -                    u64 min_block_size,
>>>> -                    unsigned long flags,
>>>> -                    struct list_head *blocks)
>>>> -{
>>>> -    u64 aligned_offset = round_down(unaligned_offset, 
>>>> min_block_size);
>>>> -
>>>> -    return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags,
>>>> -                       NULL, blocks);
>>>> -}
>>>> -
>>>>   static int __alloc_contig_try_harder(struct gpu_buddy *mm,
>>>> +                     u64 range_start, u64 range_end,
>>>>                        u64 size,
>>>>                        u64 min_block_size,
>>>>                        unsigned long flags,
>>>>                        struct list_head *blocks)
>>>>   {
>>>> -    u64 rhs_offset, lhs_offset, filled;
>>>> +    u64 rhs_offset, lhs_offset, filled, aligned;
>>>>       struct gpu_buddy_block *block;
>>>>       struct rb_root *root;
>>>>       struct rb_node *iter;
>>>> @@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct 
>>>> gpu_buddy *mm,
>>>>                              flags, &filled, blocks);
>>>>           if (err && err != -ENOSPC)
>>>>               return err;
>>>> -        if (!err && IS_ALIGNED(rhs_offset, min_block_size))
>>>> +        if (!err && IS_ALIGNED(rhs_offset, min_block_size) &&
>>>> +            rhs_offset >= range_start && rhs_offset + size <= 
>>>> range_end)
>>>>               return 0;
>>>>           if (!err) {
>>>>               /* Allocate the unaligned RHS offset using round_down */
>>>>               gpu_buddy_free_list_internal(mm, blocks);
>>>> -            err = __alloc_contig_aligned_retry(mm, rhs_offset,
>>>> -                               size,
>>>> -                               min_block_size,
>>>> -                               flags, blocks);
>>>> -            if (!err)
>>>> -                return 0;
>>>> -            if (err != -ENOSPC) {
>>>> -                gpu_buddy_free_list_internal(mm, blocks);
>>>> -                return err;
>>>> +
>>>> +            aligned = round_down(rhs_offset, min_block_size);
>>>> +            if (aligned >= range_start &&
>>>> +                aligned + size <= range_end) {
>>>> +                err = __gpu_buddy_alloc_range(mm, aligned, size,
>>>> +                                  flags, NULL, blocks);
>>>
>>> Did you consider doing a bias-range for [start, end] using 
>>> min_block_size and using what it returns as the starting point, 
>>> extending left/right? If that fails advance start and try again? 
>>> Maybe what you have here is much better for the case you have in mind?
>> I took a closer look at the alloc_range_bias approach. My 
>> understanding is that this would repeatedly bias within [start, end], 
>> use the returned min_block_size block as a starting point, and then 
>> extend left/right to build the requested run.
>> While that should work, every failed attempt may require splitting 
>> higher-order blocks to obtain a min_block_size block, followed by a 
>> free/merge back when the extension fails.
>> The free-tree descent evaluates the available starting points through 
>> a tree walk without any splitting or allocation just for enumeration, 
>> so I kept that approach.
>> Please let me know if there is a case where alloc_range_bias would 
>> provide an advantage over the free-tree descent.
>>
>> While evaluating it, I found two issues in the current try_harder() 
>> implementation:
>>
>> 1. It only searches free_tree[size_order]. This is insufficient when 
>> min_block_size < size. For example, for a 128 KiB allocation 
>> (order-5) with min_block_size = 4 KiB, free_tree[5] may be empty
>> while two adjacent order-4 (64 KiB) blocks covering [64 KiB, 128 KiB] 
>> and [128 KiB, 192 KiB] are free. These blocks still form a valid 
>> order-5-sized contiguous run, but they can never appear in
>> free_tree[5] because they are not mergeable buddies. As a result, a 
>> search that only considers free_tree[5] will never find this 
>> placement. v3 addresses this by walking the free tree from the
>> requested size order down to min_order, while reusing the existing 
>> extend/stitch logic and min_block_size alignment checks unchanged.
>>
>> 2. The search is not restricted to the requested range. v3 confines 
>> the walk to [range_start, range_end], skipping candidates at or above 
>> range_end and stopping once the walk falls below range_start.
>> This ensures that range-constrained allocations only consider free 
>> blocks within the requested window and avoids performing 
>> allocation/free attempts on blocks outside the specified range.
>> For non-range callers, the effective window remains [0, mm->size], so 
>> the behavior is unchanged.
>>
>> Regards,
>> Arun.
>>>
>>>> +                if (!err)
>>>> +                    return 0;
>>>> +                if (err != -ENOSPC) {
>>>> +                    gpu_buddy_free_list_internal(mm, blocks);
>>>> +                    return err;
>>>> +                }
>>>>               }
>>>>               goto next;
>>>>           }
>>>> @@ -1759,15 +1751,17 @@ static int __alloc_contig_try_harder(struct 
>>>> gpu_buddy *mm,
>>>>             /* Allocate the unaligned LHS offset using round_down */
>>>>           gpu_buddy_free_list_internal(mm, blocks);
>>>> -        err = __alloc_contig_aligned_retry(mm, lhs_offset,
>>>> -                           size,
>>>> -                           min_block_size,
>>>> -                           flags, blocks);
>>>> -        if (!err)
>>>> -            return 0;
>>>> -        if (err != -ENOSPC) {
>>>> -            gpu_buddy_free_list_internal(mm, blocks);
>>>> -            return err;
>>>> +
>>>> +        aligned = round_down(lhs_offset, min_block_size);
>>>> +        if (aligned >= range_start && aligned + size <= range_end) {
>>>> +            err = __gpu_buddy_alloc_range(mm, aligned, size,
>>>> +                              flags, NULL, blocks);
>>>> +            if (!err)
>>>> +                return 0;
>>>> +            if (err != -ENOSPC) {
>>>> +                gpu_buddy_free_list_internal(mm, blocks);
>>>> +                return err;
>>>> +            }
>>>>           }
>>>>   next:
>>>>           gpu_buddy_free_list_internal(mm, blocks);
>>>> @@ -2021,11 +2015,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy 
>>>> *mm,
>>>>       min_order = ilog2(min_block_size) - ilog2(mm->chunk_size);
>>>>         if (order > mm->max_order || size > mm->size) {
>>>> -        if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) &&
>>>> -            !(flags & GPU_BUDDY_RANGE_ALLOCATION))
>>>> -            return __alloc_contig_try_harder(mm, original_size,
>>>> +        if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
>>>> +            u64 range_start, range_end;
>>>> +
>>>> +            range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? 
>>>> start : 0;
>>>> +            range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end 
>>>> : mm->size;
>>>> +
>>>> +            return __alloc_contig_try_harder(mm, range_start,
>>>> +                             range_end,
>>>> +                             original_size,
>>>>                                original_min_size,
>>>>                                flags, blocks);
>>>> +        }
>>>>             return -EINVAL;
>>>>       }
>>>> @@ -2058,9 +2059,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy 
>>>> *mm,
>>>>                * Try contiguous block allocation through
>>>>                * try harder method.
>>>>                */
>>>> -            if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION &&
>>>> -                !(flags & GPU_BUDDY_RANGE_ALLOCATION)) {
>>>> -                err = __alloc_contig_try_harder(mm,
>>>> +            if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
>>>> +                u64 range_start, range_end;
>>>> +
>>>> +                range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) 
>>>> ? start : 0;
>>>> +                range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? 
>>>> end : mm->size;
>>>> +
>>>> +                err = __alloc_contig_try_harder(mm, range_start,
>>>> +                                range_end,
>>>>                                   original_size,
>>>>                                   original_min_size,
>>>>                                   flags,
>>>> @@ -2068,9 +2074,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
>>>>                   if (!err)
>>>>                       return 0;
>>>>                   if (err != -ENOSPC)
>>>> -                    return err;
>>>> -                goto err_free;
>>>> +                    goto err_free;
>>>>               }
>>>> +
>>>>               err = -ENOSPC;
>>>>               goto err_free;
>>>>           } while (1);
>>>> diff --git a/drivers/gpu/tests/gpu_buddy_test.c 
>>>> b/drivers/gpu/tests/ gpu_buddy_test.c
>>>> index b75d32ca6ca0..2c445870b808 100644
>>>> --- a/drivers/gpu/tests/gpu_buddy_test.c
>>>> +++ b/drivers/gpu/tests/gpu_buddy_test.c
>>>> @@ -1251,6 +1251,77 @@ static void 
>>>> gpu_test_buddy_alloc_contiguous(struct kunit *test)
>>>>       gpu_buddy_fini(&mm);
>>>>   }
>>>>   +static void gpu_test_buddy_alloc_range_contiguous(struct kunit 
>>>> *test)
>>>> +{
>>>> +    const unsigned long ps = SZ_4K, mm_size = 16 * ps;
>>>> +    const unsigned long range_end = 8 * ps;
>>>> +    struct gpu_buddy_block *block, *prev;
>>>> +    LIST_HEAD(allocated);
>>>> +    struct gpu_buddy mm;
>>>> +    LIST_HEAD(pin_lo);
>>>> +    LIST_HEAD(pin_hi);
>>>> +    u64 total;
>>>> +
>>>> +    KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps),
>>>> +                   "buddy_init failed\n");
>>>> +
>>>> +    /*
>>>> +     * Idea is to confine the test to the sub-range [0, 32K), 
>>>> which a 12K
>>>> +     * contiguous request (rounded up to 16K) splits into two 
>>>> naturally
>>>> +     * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the 
>>>> first 4K page
>>>> +     * of each slot ([0, 4K) and [16K, 20K)) so that neither 
>>>> aligned slot
>>>> +     * can satisfy the rounded-up 16K allocation, yet the freed 
>>>> remainder
>>>> +     * still leaves a contiguous 12K hole at offset 4K, which is 
>>>> page-aligned
>>>> +     * but not 16K-aligned. A 12K contiguous+range allocation must 
>>>> therefore
>>>> +     * fall back to stitching that span instead of returning -ENOSPC.
>>>> +     */
>>>> +    KUNIT_ASSERT_FALSE_MSG(test,
>>>> +                   gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps,
>>>> +                              &pin_lo, 0),
>>>> +                   "failed to pin low page\n");
>>>> +    KUNIT_ASSERT_FALSE_MSG(test,
>>>> +                   gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps,
>>>> +                              ps, &pin_hi, 0),
>>>> +                   "failed to pin high page\n");
>>>> +
>>>> +    /* No aligned 16K block is free; the range-aware fallback must 
>>>> stitch
>>>> +     * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC.
>>>> +     */
>>>> +    KUNIT_ASSERT_FALSE_MSG(test,
>>>> +                   gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps,
>>>> +                              ps, &allocated,
>>>> + GPU_BUDDY_CONTIGUOUS_ALLOCATION |
>>>> +                              GPU_BUDDY_RANGE_ALLOCATION),
>>>> +                   "range-restricted contiguous alloc failed\n");
>>>> +
>>>> +    /* The result must be exactly 3*ps, contiguous, and inside the 
>>>> range. */
>>>> +    total = 0;
>>>> +    prev = NULL;
>>>> +    list_for_each_entry(block, &allocated, link) {
>>>> +        u64 offset = gpu_buddy_block_offset(block);
>>>> +        u64 bsize = gpu_buddy_block_size(&mm, block);
>>>> +
>>>> +        KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end,
>>>> +                      "block [%llx, %llx) outside range\n",
>>>> +                      offset, offset + bsize);
>>>> +        if (prev)
>>>> +            KUNIT_EXPECT_EQ_MSG(test,
>>>> +                        gpu_buddy_block_offset(prev) +
>>>> +                        gpu_buddy_block_size(&mm, prev),
>>>> +                        offset,
>>>> +                        "block at %llx not contiguous\n",
>>>> +                        offset);
>>>> +        prev = block;
>>>> +        total += bsize;
>>>> +    }
>>>> +    KUNIT_EXPECT_EQ(test, total, 3 * ps);
>>>> +
>>>> +    gpu_buddy_free_list(&mm, &allocated, 0);
>>>> +    gpu_buddy_free_list(&mm, &pin_lo, 0);
>>>> +    gpu_buddy_free_list(&mm, &pin_hi, 0);
>>>> +    gpu_buddy_fini(&mm);
>>>> +}
>>>> +
>>>>   static void gpu_test_buddy_alloc_pathological(struct kunit *test)
>>>>   {
>>>>       u64 mm_size, size, start = 0;
>>>> @@ -1534,10 +1605,13 @@ static void 
>>>> gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test)
>>>>                        GPU_BUDDY_RANGE_ALLOCATION);
>>>>       KUNIT_EXPECT_EQ(test, err, -EINVAL);
>>>>   -    /* CONTIGUOUS + RANGE should return -EINVAL (no try_harder 
>>>> for RANGE) */
>>>> -    err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, 
>>>> &blocks,
>>>> -                     GPU_BUDDY_CONTIGUOUS_ALLOCATION | 
>>>> GPU_BUDDY_RANGE_ALLOCATION);
>>>> -    KUNIT_EXPECT_EQ(test, err, -EINVAL);
>>>> +    /* CONTIGUOUS + RANGE should succeed via the range-aware 
>>>> try_harder */
>>>> +    KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, 
>>>> mm_size, size,
>>>> +                                SZ_4K, &blocks,
>>>> + GPU_BUDDY_CONTIGUOUS_ALLOCATION |
>>>> + GPU_BUDDY_RANGE_ALLOCATION),
>>>> +                   "range contiguous alloc hit an error 
>>>> size=%llu\n", size);
>>>> +    gpu_buddy_free_list(&mm, &blocks, 0);
>>>>         gpu_buddy_fini(&mm);
>>>>   }
>>>> @@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = {
>>>>       KUNIT_CASE(gpu_test_buddy_alloc_pessimistic),
>>>>       KUNIT_CASE(gpu_test_buddy_alloc_pathological),
>>>>       KUNIT_CASE(gpu_test_buddy_alloc_contiguous),
>>>> +    KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous),
>>>>       KUNIT_CASE(gpu_test_buddy_alloc_clear),
>>>>       KUNIT_CASE(gpu_test_buddy_alloc_range),
>>>>       KUNIT_CASE(gpu_test_buddy_alloc_range_bias),
>>>>
>>>> base-commit: 744f262401cc2e1f3827496c72a47e072d31a852
>>>
>>
>


^ permalink raw reply	[flat|nested] 8+ messages in thread

* Re: [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback
  2026-10-09  9:14       ` Arunpravin Paneer Selvam
@ 2026-10-09 10:17         ` Matthew Auld
  0 siblings, 0 replies; 8+ messages in thread
From: Matthew Auld @ 2026-10-09 10:17 UTC (permalink / raw)
  To: Arunpravin Paneer Selvam, dri-devel, intel-gfx, intel-xe, amd-gfx
  Cc: christian.koenig, alexander.deucher, Anand.Raghavendra

On 09/10/2026 10:14, Arunpravin Paneer Selvam wrote:
> 
> 
> On 10/6/2026 3:27 PM, Matthew Auld wrote:
>> On 05/10/2026 15:14, Arunpravin Paneer Selvam wrote:
>>>
>>>
>>> On 10/1/2026 11:58 PM, Matthew Auld wrote:
>>>> On 30/09/2026 07:21, Arunpravin Paneer Selvam wrote:
>>>>> From: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com>
>>>>>
>>>>> A range + contiguous allocation (e.g. a scanout FB confined to the
>>>>> CPU-visible VRAM aperture) rounds its size up to a power of two and
>>>>> requires a naturally aligned free block of that size; on a fragmented
>>>>> aperture no such aligned block may exist even though enough contiguous
>>>>> space is free, so the allocation fails with -ENOSPC.
>>>>>
>>>>> The non-range contiguous path already recovers from this via
>>>>> __alloc_contig_try_harder(), which stitches an exact-size span from
>>>>> smaller adjacent blocks, but that fallback was unreachable once
>>>>> GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder()
>>>>> a [range_start, range_end) window and route the range + contiguous
>>>>> case through it: each candidate placement is confined to the window
>>>>> and aligned to min_block_size. Non-range callers pass [0, mm->size),
>>>>> where the guards are no-ops, so the existing behaviour is unchanged.
>>>>> A KUnit regression test covers the fragmentation pattern.
>>>>>
>>>>> Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b
>>>>> regression.
>>>>>
>>>>> v2:
>>>>>   - Drop the split-undo patch; the range-bias search always 
>>>>> descends to
>>>>>     an exact-order block or fails the split (already handled), so the
>>>>>     extra undo was redundant. (Matthew)
>>>>>   - Verified this fallback alone fixes the kms_plane regression.
>>>>>
>>>>> Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with 
>>>>> decoupled dirty tracker")
>>>>
>>>> Patch looks more like totally new functionally/feature, so the fixes 
>>>> here is maybe unexpected? Did something in that fixes commit change 
>>>> something such that the try_harder is now needed, but that needs 
>>>> some expansion with bias + contig? Do we know exactly what changed 
>>>> here?
>>> That commit removed __force_merge(), which previously helped recover 
>>> larger contiguous allocations on demand. As a result, some range- 
>>
>> __force_merge() got nuked, but IIRC I think that was essentially 
>> because we now "force merge" on free, so shouldn't we get the ~same 
>> result? Or is the fact that we force_merge() on every free giving a 
>> different layout of pages, and with some very some specific allocation 
>> pattern that difference subtly results in -ENOSPC, somehow?
> You are right - merge-on-free reclaims contiguity just like the old on- 
> demand  __force_merge() , so that is not the cause and the layout is 
> effectively the same. The real change is where clear/dirty segregation 
> lives: it used to be an inline guard inside __alloc_range_bias() , but 
> the rework pulled it out into dirty_steer_window() , which only runs for 
> non-range allocations. So range (aperture) requests now go through 
> __alloc_range_bias()  with no clear/dirty guard at all - clear and dirty 
> allocations interleave and fragment the aperture until no large same- 
> class run survives for a 4K scanout pin (-> -ENOSPC). Restoring that 
> two-pass guard is the real fix.

Ahh, yeah that makes more sense now :)

> 
> I have accordingly moved the  Fixes:  tag onto that guard-restore patch, 
> and kept the range support in  __alloc_contig_try_harder() as a new 
> feature -  it still helps, but it is a safety-net fallback that should 
> land on top, not as the root-cause fix.
> 
> Regards,
> Arun.
>>
>>> restricted contiguous allocation requests
>>> that used to succeed now fail with -ENOSPC (the kms_plane 
>>> regression). This patch restores that capability through the 
>>> try_harder() stitching logic, so the Fixes: tag reflects a
>>> regression introduced by that commit rather than new functionality.
>>>>
>>>>> Assisted-by: Claude:claude-opus-4-8
>>>>
>>>> Assisted-by: LLM
>>>>
>>>>> Cc: Matthew Auld <matthew.auld@intel.com>
>>>>> Cc: Christian König <christian.koenig@amd.com>
>>>>> Signed-off-by: Arunpravin Paneer Selvam 
>>>>> <Arunpravin.PaneerSelvam@amd.com>
>>>>> ---
>>>>>   drivers/gpu/buddy.c                | 88 +++++++++++++++ 
>>>>> +--------------
>>>>>   drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++--
>>>>>   2 files changed, 126 insertions(+), 45 deletions(-)
>>>>>
>>>>> diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c
>>>>> index 2f2aaadafe35..5265e1f6a318 100644
>>>>> --- a/drivers/gpu/buddy.c
>>>>> +++ b/drivers/gpu/buddy.c
>>>>> @@ -1685,26 +1685,14 @@ static int __gpu_buddy_alloc_range(struct 
>>>>> gpu_buddy *mm,
>>>>>                    blocks, total_allocated_on_err);
>>>>>   }
>>>>>   -static int __alloc_contig_aligned_retry(struct gpu_buddy *mm,
>>>>> -                    u64 unaligned_offset,
>>>>> -                    u64 size,
>>>>> -                    u64 min_block_size,
>>>>> -                    unsigned long flags,
>>>>> -                    struct list_head *blocks)
>>>>> -{
>>>>> -    u64 aligned_offset = round_down(unaligned_offset, 
>>>>> min_block_size);
>>>>> -
>>>>> -    return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags,
>>>>> -                       NULL, blocks);
>>>>> -}
>>>>> -
>>>>>   static int __alloc_contig_try_harder(struct gpu_buddy *mm,
>>>>> +                     u64 range_start, u64 range_end,
>>>>>                        u64 size,
>>>>>                        u64 min_block_size,
>>>>>                        unsigned long flags,
>>>>>                        struct list_head *blocks)
>>>>>   {
>>>>> -    u64 rhs_offset, lhs_offset, filled;
>>>>> +    u64 rhs_offset, lhs_offset, filled, aligned;
>>>>>       struct gpu_buddy_block *block;
>>>>>       struct rb_root *root;
>>>>>       struct rb_node *iter;
>>>>> @@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct 
>>>>> gpu_buddy *mm,
>>>>>                              flags, &filled, blocks);
>>>>>           if (err && err != -ENOSPC)
>>>>>               return err;
>>>>> -        if (!err && IS_ALIGNED(rhs_offset, min_block_size))
>>>>> +        if (!err && IS_ALIGNED(rhs_offset, min_block_size) &&
>>>>> +            rhs_offset >= range_start && rhs_offset + size <= 
>>>>> range_end)
>>>>>               return 0;
>>>>>           if (!err) {
>>>>>               /* Allocate the unaligned RHS offset using round_down */
>>>>>               gpu_buddy_free_list_internal(mm, blocks);
>>>>> -            err = __alloc_contig_aligned_retry(mm, rhs_offset,
>>>>> -                               size,
>>>>> -                               min_block_size,
>>>>> -                               flags, blocks);
>>>>> -            if (!err)
>>>>> -                return 0;
>>>>> -            if (err != -ENOSPC) {
>>>>> -                gpu_buddy_free_list_internal(mm, blocks);
>>>>> -                return err;
>>>>> +
>>>>> +            aligned = round_down(rhs_offset, min_block_size);
>>>>> +            if (aligned >= range_start &&
>>>>> +                aligned + size <= range_end) {
>>>>> +                err = __gpu_buddy_alloc_range(mm, aligned, size,
>>>>> +                                  flags, NULL, blocks);
>>>>
>>>> Did you consider doing a bias-range for [start, end] using 
>>>> min_block_size and using what it returns as the starting point, 
>>>> extending left/right? If that fails advance start and try again? 
>>>> Maybe what you have here is much better for the case you have in mind?
>>> I took a closer look at the alloc_range_bias approach. My 
>>> understanding is that this would repeatedly bias within [start, end], 
>>> use the returned min_block_size block as a starting point, and then 
>>> extend left/right to build the requested run.
>>> While that should work, every failed attempt may require splitting 
>>> higher-order blocks to obtain a min_block_size block, followed by a 
>>> free/merge back when the extension fails.
>>> The free-tree descent evaluates the available starting points through 
>>> a tree walk without any splitting or allocation just for enumeration, 
>>> so I kept that approach.
>>> Please let me know if there is a case where alloc_range_bias would 
>>> provide an advantage over the free-tree descent.
>>>
>>> While evaluating it, I found two issues in the current try_harder() 
>>> implementation:
>>>
>>> 1. It only searches free_tree[size_order]. This is insufficient when 
>>> min_block_size < size. For example, for a 128 KiB allocation 
>>> (order-5) with min_block_size = 4 KiB, free_tree[5] may be empty
>>> while two adjacent order-4 (64 KiB) blocks covering [64 KiB, 128 KiB] 
>>> and [128 KiB, 192 KiB] are free. These blocks still form a valid 
>>> order-5-sized contiguous run, but they can never appear in
>>> free_tree[5] because they are not mergeable buddies. As a result, a 
>>> search that only considers free_tree[5] will never find this 
>>> placement. v3 addresses this by walking the free tree from the
>>> requested size order down to min_order, while reusing the existing 
>>> extend/stitch logic and min_block_size alignment checks unchanged.
>>>
>>> 2. The search is not restricted to the requested range. v3 confines 
>>> the walk to [range_start, range_end], skipping candidates at or above 
>>> range_end and stopping once the walk falls below range_start.
>>> This ensures that range-constrained allocations only consider free 
>>> blocks within the requested window and avoids performing allocation/ 
>>> free attempts on blocks outside the specified range.
>>> For non-range callers, the effective window remains [0, mm->size], so 
>>> the behavior is unchanged.
>>>
>>> Regards,
>>> Arun.
>>>>
>>>>> +                if (!err)
>>>>> +                    return 0;
>>>>> +                if (err != -ENOSPC) {
>>>>> +                    gpu_buddy_free_list_internal(mm, blocks);
>>>>> +                    return err;
>>>>> +                }
>>>>>               }
>>>>>               goto next;
>>>>>           }
>>>>> @@ -1759,15 +1751,17 @@ static int __alloc_contig_try_harder(struct 
>>>>> gpu_buddy *mm,
>>>>>             /* Allocate the unaligned LHS offset using round_down */
>>>>>           gpu_buddy_free_list_internal(mm, blocks);
>>>>> -        err = __alloc_contig_aligned_retry(mm, lhs_offset,
>>>>> -                           size,
>>>>> -                           min_block_size,
>>>>> -                           flags, blocks);
>>>>> -        if (!err)
>>>>> -            return 0;
>>>>> -        if (err != -ENOSPC) {
>>>>> -            gpu_buddy_free_list_internal(mm, blocks);
>>>>> -            return err;
>>>>> +
>>>>> +        aligned = round_down(lhs_offset, min_block_size);
>>>>> +        if (aligned >= range_start && aligned + size <= range_end) {
>>>>> +            err = __gpu_buddy_alloc_range(mm, aligned, size,
>>>>> +                              flags, NULL, blocks);
>>>>> +            if (!err)
>>>>> +                return 0;
>>>>> +            if (err != -ENOSPC) {
>>>>> +                gpu_buddy_free_list_internal(mm, blocks);
>>>>> +                return err;
>>>>> +            }
>>>>>           }
>>>>>   next:
>>>>>           gpu_buddy_free_list_internal(mm, blocks);
>>>>> @@ -2021,11 +2015,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy 
>>>>> *mm,
>>>>>       min_order = ilog2(min_block_size) - ilog2(mm->chunk_size);
>>>>>         if (order > mm->max_order || size > mm->size) {
>>>>> -        if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) &&
>>>>> -            !(flags & GPU_BUDDY_RANGE_ALLOCATION))
>>>>> -            return __alloc_contig_try_harder(mm, original_size,
>>>>> +        if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
>>>>> +            u64 range_start, range_end;
>>>>> +
>>>>> +            range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? 
>>>>> start : 0;
>>>>> +            range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? 
>>>>> end : mm->size;
>>>>> +
>>>>> +            return __alloc_contig_try_harder(mm, range_start,
>>>>> +                             range_end,
>>>>> +                             original_size,
>>>>>                                original_min_size,
>>>>>                                flags, blocks);
>>>>> +        }
>>>>>             return -EINVAL;
>>>>>       }
>>>>> @@ -2058,9 +2059,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy 
>>>>> *mm,
>>>>>                * Try contiguous block allocation through
>>>>>                * try harder method.
>>>>>                */
>>>>> -            if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION &&
>>>>> -                !(flags & GPU_BUDDY_RANGE_ALLOCATION)) {
>>>>> -                err = __alloc_contig_try_harder(mm,
>>>>> +            if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
>>>>> +                u64 range_start, range_end;
>>>>> +
>>>>> +                range_start = (flags & 
>>>>> GPU_BUDDY_RANGE_ALLOCATION) ? start : 0;
>>>>> +                range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? 
>>>>> end : mm->size;
>>>>> +
>>>>> +                err = __alloc_contig_try_harder(mm, range_start,
>>>>> +                                range_end,
>>>>>                                   original_size,
>>>>>                                   original_min_size,
>>>>>                                   flags,
>>>>> @@ -2068,9 +2074,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
>>>>>                   if (!err)
>>>>>                       return 0;
>>>>>                   if (err != -ENOSPC)
>>>>> -                    return err;
>>>>> -                goto err_free;
>>>>> +                    goto err_free;
>>>>>               }
>>>>> +
>>>>>               err = -ENOSPC;
>>>>>               goto err_free;
>>>>>           } while (1);
>>>>> diff --git a/drivers/gpu/tests/gpu_buddy_test.c b/drivers/gpu/ 
>>>>> tests/ gpu_buddy_test.c
>>>>> index b75d32ca6ca0..2c445870b808 100644
>>>>> --- a/drivers/gpu/tests/gpu_buddy_test.c
>>>>> +++ b/drivers/gpu/tests/gpu_buddy_test.c
>>>>> @@ -1251,6 +1251,77 @@ static void 
>>>>> gpu_test_buddy_alloc_contiguous(struct kunit *test)
>>>>>       gpu_buddy_fini(&mm);
>>>>>   }
>>>>>   +static void gpu_test_buddy_alloc_range_contiguous(struct kunit 
>>>>> *test)
>>>>> +{
>>>>> +    const unsigned long ps = SZ_4K, mm_size = 16 * ps;
>>>>> +    const unsigned long range_end = 8 * ps;
>>>>> +    struct gpu_buddy_block *block, *prev;
>>>>> +    LIST_HEAD(allocated);
>>>>> +    struct gpu_buddy mm;
>>>>> +    LIST_HEAD(pin_lo);
>>>>> +    LIST_HEAD(pin_hi);
>>>>> +    u64 total;
>>>>> +
>>>>> +    KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps),
>>>>> +                   "buddy_init failed\n");
>>>>> +
>>>>> +    /*
>>>>> +     * Idea is to confine the test to the sub-range [0, 32K), 
>>>>> which a 12K
>>>>> +     * contiguous request (rounded up to 16K) splits into two 
>>>>> naturally
>>>>> +     * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the 
>>>>> first 4K page
>>>>> +     * of each slot ([0, 4K) and [16K, 20K)) so that neither 
>>>>> aligned slot
>>>>> +     * can satisfy the rounded-up 16K allocation, yet the freed 
>>>>> remainder
>>>>> +     * still leaves a contiguous 12K hole at offset 4K, which is 
>>>>> page-aligned
>>>>> +     * but not 16K-aligned. A 12K contiguous+range allocation must 
>>>>> therefore
>>>>> +     * fall back to stitching that span instead of returning -ENOSPC.
>>>>> +     */
>>>>> +    KUNIT_ASSERT_FALSE_MSG(test,
>>>>> +                   gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps,
>>>>> +                              &pin_lo, 0),
>>>>> +                   "failed to pin low page\n");
>>>>> +    KUNIT_ASSERT_FALSE_MSG(test,
>>>>> +                   gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps,
>>>>> +                              ps, &pin_hi, 0),
>>>>> +                   "failed to pin high page\n");
>>>>> +
>>>>> +    /* No aligned 16K block is free; the range-aware fallback must 
>>>>> stitch
>>>>> +     * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC.
>>>>> +     */
>>>>> +    KUNIT_ASSERT_FALSE_MSG(test,
>>>>> +                   gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps,
>>>>> +                              ps, &allocated,
>>>>> + GPU_BUDDY_CONTIGUOUS_ALLOCATION |
>>>>> +                              GPU_BUDDY_RANGE_ALLOCATION),
>>>>> +                   "range-restricted contiguous alloc failed\n");
>>>>> +
>>>>> +    /* The result must be exactly 3*ps, contiguous, and inside the 
>>>>> range. */
>>>>> +    total = 0;
>>>>> +    prev = NULL;
>>>>> +    list_for_each_entry(block, &allocated, link) {
>>>>> +        u64 offset = gpu_buddy_block_offset(block);
>>>>> +        u64 bsize = gpu_buddy_block_size(&mm, block);
>>>>> +
>>>>> +        KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end,
>>>>> +                      "block [%llx, %llx) outside range\n",
>>>>> +                      offset, offset + bsize);
>>>>> +        if (prev)
>>>>> +            KUNIT_EXPECT_EQ_MSG(test,
>>>>> +                        gpu_buddy_block_offset(prev) +
>>>>> +                        gpu_buddy_block_size(&mm, prev),
>>>>> +                        offset,
>>>>> +                        "block at %llx not contiguous\n",
>>>>> +                        offset);
>>>>> +        prev = block;
>>>>> +        total += bsize;
>>>>> +    }
>>>>> +    KUNIT_EXPECT_EQ(test, total, 3 * ps);
>>>>> +
>>>>> +    gpu_buddy_free_list(&mm, &allocated, 0);
>>>>> +    gpu_buddy_free_list(&mm, &pin_lo, 0);
>>>>> +    gpu_buddy_free_list(&mm, &pin_hi, 0);
>>>>> +    gpu_buddy_fini(&mm);
>>>>> +}
>>>>> +
>>>>>   static void gpu_test_buddy_alloc_pathological(struct kunit *test)
>>>>>   {
>>>>>       u64 mm_size, size, start = 0;
>>>>> @@ -1534,10 +1605,13 @@ static void 
>>>>> gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test)
>>>>>                        GPU_BUDDY_RANGE_ALLOCATION);
>>>>>       KUNIT_EXPECT_EQ(test, err, -EINVAL);
>>>>>   -    /* CONTIGUOUS + RANGE should return -EINVAL (no try_harder 
>>>>> for RANGE) */
>>>>> -    err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, 
>>>>> &blocks,
>>>>> -                     GPU_BUDDY_CONTIGUOUS_ALLOCATION | 
>>>>> GPU_BUDDY_RANGE_ALLOCATION);
>>>>> -    KUNIT_EXPECT_EQ(test, err, -EINVAL);
>>>>> +    /* CONTIGUOUS + RANGE should succeed via the range-aware 
>>>>> try_harder */
>>>>> +    KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, 
>>>>> mm_size, size,
>>>>> +                                SZ_4K, &blocks,
>>>>> + GPU_BUDDY_CONTIGUOUS_ALLOCATION |
>>>>> + GPU_BUDDY_RANGE_ALLOCATION),
>>>>> +                   "range contiguous alloc hit an error 
>>>>> size=%llu\n", size);
>>>>> +    gpu_buddy_free_list(&mm, &blocks, 0);
>>>>>         gpu_buddy_fini(&mm);
>>>>>   }
>>>>> @@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = {
>>>>>       KUNIT_CASE(gpu_test_buddy_alloc_pessimistic),
>>>>>       KUNIT_CASE(gpu_test_buddy_alloc_pathological),
>>>>>       KUNIT_CASE(gpu_test_buddy_alloc_contiguous),
>>>>> +    KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous),
>>>>>       KUNIT_CASE(gpu_test_buddy_alloc_clear),
>>>>>       KUNIT_CASE(gpu_test_buddy_alloc_range),
>>>>>       KUNIT_CASE(gpu_test_buddy_alloc_range_bias),
>>>>>
>>>>> base-commit: 744f262401cc2e1f3827496c72a47e072d31a852
>>>>
>>>
>>
> 


^ permalink raw reply	[flat|nested] 8+ messages in thread

end of thread, other threads:[~2026-10-09 10:17 UTC | newest]

Thread overview: 8+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-09-30  6:21 [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback Arunpravin Paneer Selvam
2026-09-30  6:31 ` sashiko-bot
2026-10-01  7:06 ` Arunpravin Paneer Selvam
2026-10-01 18:28 ` Matthew Auld
2026-10-05 14:14   ` Arunpravin Paneer Selvam
2026-10-06  9:57     ` Matthew Auld
2026-10-09  9:14       ` Arunpravin Paneer Selvam
2026-10-09 10:17         ` Matthew Auld

This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox