* [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback
@ 2026-09-30 6:21 Arunpravin Paneer Selvam
2026-09-30 6:31 ` sashiko-bot
` (2 more replies)
0 siblings, 3 replies; 8+ messages in thread
From: Arunpravin Paneer Selvam @ 2026-09-30 6:21 UTC (permalink / raw)
To: matthew.auld, dri-devel, intel-gfx, intel-xe, amd-gfx
Cc: christian.koenig, alexander.deucher, Anand.Raghavendra,
Arunpravin Paneer Selvam
From: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com>
A range + contiguous allocation (e.g. a scanout FB confined to the
CPU-visible VRAM aperture) rounds its size up to a power of two and
requires a naturally aligned free block of that size; on a fragmented
aperture no such aligned block may exist even though enough contiguous
space is free, so the allocation fails with -ENOSPC.
The non-range contiguous path already recovers from this via
__alloc_contig_try_harder(), which stitches an exact-size span from
smaller adjacent blocks, but that fallback was unreachable once
GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder()
a [range_start, range_end) window and route the range + contiguous
case through it: each candidate placement is confined to the window
and aligned to min_block_size. Non-range callers pass [0, mm->size),
where the guards are no-ops, so the existing behaviour is unchanged.
A KUnit regression test covers the fragmentation pattern.
Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b
regression.
v2:
- Drop the split-undo patch; the range-bias search always descends to
an exact-order block or fails the split (already handled), so the
extra undo was redundant. (Matthew)
- Verified this fallback alone fixes the kms_plane regression.
Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with decoupled dirty tracker")
Assisted-by: Claude:claude-opus-4-8
Cc: Matthew Auld <matthew.auld@intel.com>
Cc: Christian König <christian.koenig@amd.com>
Signed-off-by: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com>
---
drivers/gpu/buddy.c | 88 ++++++++++++++++--------------
drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++--
2 files changed, 126 insertions(+), 45 deletions(-)
diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c
index 2f2aaadafe35..5265e1f6a318 100644
--- a/drivers/gpu/buddy.c
+++ b/drivers/gpu/buddy.c
@@ -1685,26 +1685,14 @@ static int __gpu_buddy_alloc_range(struct gpu_buddy *mm,
blocks, total_allocated_on_err);
}
-static int __alloc_contig_aligned_retry(struct gpu_buddy *mm,
- u64 unaligned_offset,
- u64 size,
- u64 min_block_size,
- unsigned long flags,
- struct list_head *blocks)
-{
- u64 aligned_offset = round_down(unaligned_offset, min_block_size);
-
- return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags,
- NULL, blocks);
-}
-
static int __alloc_contig_try_harder(struct gpu_buddy *mm,
+ u64 range_start, u64 range_end,
u64 size,
u64 min_block_size,
unsigned long flags,
struct list_head *blocks)
{
- u64 rhs_offset, lhs_offset, filled;
+ u64 rhs_offset, lhs_offset, filled, aligned;
struct gpu_buddy_block *block;
struct rb_root *root;
struct rb_node *iter;
@@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct gpu_buddy *mm,
flags, &filled, blocks);
if (err && err != -ENOSPC)
return err;
- if (!err && IS_ALIGNED(rhs_offset, min_block_size))
+ if (!err && IS_ALIGNED(rhs_offset, min_block_size) &&
+ rhs_offset >= range_start && rhs_offset + size <= range_end)
return 0;
if (!err) {
/* Allocate the unaligned RHS offset using round_down */
gpu_buddy_free_list_internal(mm, blocks);
- err = __alloc_contig_aligned_retry(mm, rhs_offset,
- size,
- min_block_size,
- flags, blocks);
- if (!err)
- return 0;
- if (err != -ENOSPC) {
- gpu_buddy_free_list_internal(mm, blocks);
- return err;
+
+ aligned = round_down(rhs_offset, min_block_size);
+ if (aligned >= range_start &&
+ aligned + size <= range_end) {
+ err = __gpu_buddy_alloc_range(mm, aligned, size,
+ flags, NULL, blocks);
+ if (!err)
+ return 0;
+ if (err != -ENOSPC) {
+ gpu_buddy_free_list_internal(mm, blocks);
+ return err;
+ }
}
goto next;
}
@@ -1759,15 +1751,17 @@ static int __alloc_contig_try_harder(struct gpu_buddy *mm,
/* Allocate the unaligned LHS offset using round_down */
gpu_buddy_free_list_internal(mm, blocks);
- err = __alloc_contig_aligned_retry(mm, lhs_offset,
- size,
- min_block_size,
- flags, blocks);
- if (!err)
- return 0;
- if (err != -ENOSPC) {
- gpu_buddy_free_list_internal(mm, blocks);
- return err;
+
+ aligned = round_down(lhs_offset, min_block_size);
+ if (aligned >= range_start && aligned + size <= range_end) {
+ err = __gpu_buddy_alloc_range(mm, aligned, size,
+ flags, NULL, blocks);
+ if (!err)
+ return 0;
+ if (err != -ENOSPC) {
+ gpu_buddy_free_list_internal(mm, blocks);
+ return err;
+ }
}
next:
gpu_buddy_free_list_internal(mm, blocks);
@@ -2021,11 +2015,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
min_order = ilog2(min_block_size) - ilog2(mm->chunk_size);
if (order > mm->max_order || size > mm->size) {
- if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) &&
- !(flags & GPU_BUDDY_RANGE_ALLOCATION))
- return __alloc_contig_try_harder(mm, original_size,
+ if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
+ u64 range_start, range_end;
+
+ range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? start : 0;
+ range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : mm->size;
+
+ return __alloc_contig_try_harder(mm, range_start,
+ range_end,
+ original_size,
original_min_size,
flags, blocks);
+ }
return -EINVAL;
}
@@ -2058,9 +2059,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
* Try contiguous block allocation through
* try harder method.
*/
- if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION &&
- !(flags & GPU_BUDDY_RANGE_ALLOCATION)) {
- err = __alloc_contig_try_harder(mm,
+ if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
+ u64 range_start, range_end;
+
+ range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? start : 0;
+ range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : mm->size;
+
+ err = __alloc_contig_try_harder(mm, range_start,
+ range_end,
original_size,
original_min_size,
flags,
@@ -2068,9 +2074,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
if (!err)
return 0;
if (err != -ENOSPC)
- return err;
- goto err_free;
+ goto err_free;
}
+
err = -ENOSPC;
goto err_free;
} while (1);
diff --git a/drivers/gpu/tests/gpu_buddy_test.c b/drivers/gpu/tests/gpu_buddy_test.c
index b75d32ca6ca0..2c445870b808 100644
--- a/drivers/gpu/tests/gpu_buddy_test.c
+++ b/drivers/gpu/tests/gpu_buddy_test.c
@@ -1251,6 +1251,77 @@ static void gpu_test_buddy_alloc_contiguous(struct kunit *test)
gpu_buddy_fini(&mm);
}
+static void gpu_test_buddy_alloc_range_contiguous(struct kunit *test)
+{
+ const unsigned long ps = SZ_4K, mm_size = 16 * ps;
+ const unsigned long range_end = 8 * ps;
+ struct gpu_buddy_block *block, *prev;
+ LIST_HEAD(allocated);
+ struct gpu_buddy mm;
+ LIST_HEAD(pin_lo);
+ LIST_HEAD(pin_hi);
+ u64 total;
+
+ KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps),
+ "buddy_init failed\n");
+
+ /*
+ * Idea is to confine the test to the sub-range [0, 32K), which a 12K
+ * contiguous request (rounded up to 16K) splits into two naturally
+ * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the first 4K page
+ * of each slot ([0, 4K) and [16K, 20K)) so that neither aligned slot
+ * can satisfy the rounded-up 16K allocation, yet the freed remainder
+ * still leaves a contiguous 12K hole at offset 4K, which is page-aligned
+ * but not 16K-aligned. A 12K contiguous+range allocation must therefore
+ * fall back to stitching that span instead of returning -ENOSPC.
+ */
+ KUNIT_ASSERT_FALSE_MSG(test,
+ gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps,
+ &pin_lo, 0),
+ "failed to pin low page\n");
+ KUNIT_ASSERT_FALSE_MSG(test,
+ gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps,
+ ps, &pin_hi, 0),
+ "failed to pin high page\n");
+
+ /* No aligned 16K block is free; the range-aware fallback must stitch
+ * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC.
+ */
+ KUNIT_ASSERT_FALSE_MSG(test,
+ gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps,
+ ps, &allocated,
+ GPU_BUDDY_CONTIGUOUS_ALLOCATION |
+ GPU_BUDDY_RANGE_ALLOCATION),
+ "range-restricted contiguous alloc failed\n");
+
+ /* The result must be exactly 3*ps, contiguous, and inside the range. */
+ total = 0;
+ prev = NULL;
+ list_for_each_entry(block, &allocated, link) {
+ u64 offset = gpu_buddy_block_offset(block);
+ u64 bsize = gpu_buddy_block_size(&mm, block);
+
+ KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end,
+ "block [%llx, %llx) outside range\n",
+ offset, offset + bsize);
+ if (prev)
+ KUNIT_EXPECT_EQ_MSG(test,
+ gpu_buddy_block_offset(prev) +
+ gpu_buddy_block_size(&mm, prev),
+ offset,
+ "block at %llx not contiguous\n",
+ offset);
+ prev = block;
+ total += bsize;
+ }
+ KUNIT_EXPECT_EQ(test, total, 3 * ps);
+
+ gpu_buddy_free_list(&mm, &allocated, 0);
+ gpu_buddy_free_list(&mm, &pin_lo, 0);
+ gpu_buddy_free_list(&mm, &pin_hi, 0);
+ gpu_buddy_fini(&mm);
+}
+
static void gpu_test_buddy_alloc_pathological(struct kunit *test)
{
u64 mm_size, size, start = 0;
@@ -1534,10 +1605,13 @@ static void gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test)
GPU_BUDDY_RANGE_ALLOCATION);
KUNIT_EXPECT_EQ(test, err, -EINVAL);
- /* CONTIGUOUS + RANGE should return -EINVAL (no try_harder for RANGE) */
- err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, &blocks,
- GPU_BUDDY_CONTIGUOUS_ALLOCATION | GPU_BUDDY_RANGE_ALLOCATION);
- KUNIT_EXPECT_EQ(test, err, -EINVAL);
+ /* CONTIGUOUS + RANGE should succeed via the range-aware try_harder */
+ KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, mm_size, size,
+ SZ_4K, &blocks,
+ GPU_BUDDY_CONTIGUOUS_ALLOCATION |
+ GPU_BUDDY_RANGE_ALLOCATION),
+ "range contiguous alloc hit an error size=%llu\n", size);
+ gpu_buddy_free_list(&mm, &blocks, 0);
gpu_buddy_fini(&mm);
}
@@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = {
KUNIT_CASE(gpu_test_buddy_alloc_pessimistic),
KUNIT_CASE(gpu_test_buddy_alloc_pathological),
KUNIT_CASE(gpu_test_buddy_alloc_contiguous),
+ KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous),
KUNIT_CASE(gpu_test_buddy_alloc_clear),
KUNIT_CASE(gpu_test_buddy_alloc_range),
KUNIT_CASE(gpu_test_buddy_alloc_range_bias),
base-commit: 744f262401cc2e1f3827496c72a47e072d31a852
--
2.43.0
^ permalink raw reply related [flat|nested] 8+ messages in thread* Re: [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback 2026-09-30 6:21 [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback Arunpravin Paneer Selvam @ 2026-09-30 6:31 ` sashiko-bot 2026-10-01 7:06 ` Arunpravin Paneer Selvam 2026-10-01 18:28 ` Matthew Auld 2 siblings, 0 replies; 8+ messages in thread From: sashiko-bot @ 2026-09-30 6:31 UTC (permalink / raw) To: Arunpravin Paneer Selvam; +Cc: dri-devel, intel-xe, intel-gfx Thank you for your contribution! Sashiko AI review found 1 potential issue(s) to consider: - [High] gpu/buddy: range-restricted alloc fails when free space exceeds range_end -- commit 2ed4c9d1a70c51390968b65e939095ce52c8ee07 Author: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com> gpu/buddy: add range-restricted contiguous allocation fallback A range + contiguous allocation rounds its size up to a power of two and requires a naturally aligned free block of that size. The non-range contiguous path already recovers from fragmentation via __alloc_contig_try_harder(), which is now given a range window to support the range + contiguous case. > diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c > index 2f2aaadafe351..5265e1f6a3183 100644 > --- a/drivers/gpu/buddy.c > +++ b/drivers/gpu/buddy.c [ ... ] > @@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct gpu_buddy *mm, [ ... ] > if (!err) { > /* Allocate the unaligned RHS offset using round_down */ > gpu_buddy_free_list_internal(mm, blocks); > - err = __alloc_contig_aligned_retry(mm, rhs_offset, > - size, > - min_block_size, > - flags, blocks); > - if (!err) > - return 0; > - if (err != -ENOSPC) { > - gpu_buddy_free_list_internal(mm, blocks); > - return err; > + > + aligned = round_down(rhs_offset, min_block_size); > + if (aligned >= range_start && > + aligned + size <= range_end) { > + err = __gpu_buddy_alloc_range(mm, aligned, size, > + flags, NULL, blocks); > + if (!err) > + return 0; > + if (err != -ENOSPC) { > + gpu_buddy_free_list_internal(mm, blocks); > + return err; > + } > } > goto next; > } [Severity: High] Does this skip the left-hand side (LHS) fallback when contiguous free space exceeds range_end? If __gpu_buddy_alloc_range() succeeds at finding free space starting at rhs_offset, err is 0. If the free space extends beyond range_end, the newly added check (aligned >= range_start && aligned + size <= range_end) might fail. But since err is still 0, it hits the unconditional goto next. By skipping to next, doesn't this bypass the LHS window-sliding logic below? If there was valid free space to the left of rhs_offset that perfectly fits the window, it appears it would be ignored, causing the allocator to return -ENOSPC despite sufficient space existing within the requested range. -- Sashiko AI review · https://sashiko.dev/#/patchset/20260930062117.585089-1-arunpravin.paneerselvam@amd.com?part=1 ^ permalink raw reply [flat|nested] 8+ messages in thread
* Re: [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback 2026-09-30 6:21 [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback Arunpravin Paneer Selvam 2026-09-30 6:31 ` sashiko-bot @ 2026-10-01 7:06 ` Arunpravin Paneer Selvam 2026-10-01 18:28 ` Matthew Auld 2 siblings, 0 replies; 8+ messages in thread From: Arunpravin Paneer Selvam @ 2026-10-01 7:06 UTC (permalink / raw) To: matthew.auld, dri-devel, intel-gfx, intel-xe, amd-gfx Cc: christian.koenig, alexander.deucher, Anand.Raghavendra Hi Matthew, Could you please review this patch. Thanks, Arun. On 9/30/2026 11:51 AM, Arunpravin Paneer Selvam wrote: > From: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com> > > A range + contiguous allocation (e.g. a scanout FB confined to the > CPU-visible VRAM aperture) rounds its size up to a power of two and > requires a naturally aligned free block of that size; on a fragmented > aperture no such aligned block may exist even though enough contiguous > space is free, so the allocation fails with -ENOSPC. > > The non-range contiguous path already recovers from this via > __alloc_contig_try_harder(), which stitches an exact-size span from > smaller adjacent blocks, but that fallback was unreachable once > GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder() > a [range_start, range_end) window and route the range + contiguous > case through it: each candidate placement is confined to the window > and aligned to min_block_size. Non-range callers pass [0, mm->size), > where the guards are no-ops, so the existing behaviour is unchanged. > A KUnit regression test covers the fragmentation pattern. > > Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b > regression. > > v2: > - Drop the split-undo patch; the range-bias search always descends to > an exact-order block or fails the split (already handled), so the > extra undo was redundant. (Matthew) > - Verified this fallback alone fixes the kms_plane regression. > > Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with decoupled dirty tracker") > Assisted-by: Claude:claude-opus-4-8 > Cc: Matthew Auld <matthew.auld@intel.com> > Cc: Christian König <christian.koenig@amd.com> > Signed-off-by: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com> > --- > drivers/gpu/buddy.c | 88 ++++++++++++++++-------------- > drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++-- > 2 files changed, 126 insertions(+), 45 deletions(-) > > diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c > index 2f2aaadafe35..5265e1f6a318 100644 > --- a/drivers/gpu/buddy.c > +++ b/drivers/gpu/buddy.c > @@ -1685,26 +1685,14 @@ static int __gpu_buddy_alloc_range(struct gpu_buddy *mm, > blocks, total_allocated_on_err); > } > > -static int __alloc_contig_aligned_retry(struct gpu_buddy *mm, > - u64 unaligned_offset, > - u64 size, > - u64 min_block_size, > - unsigned long flags, > - struct list_head *blocks) > -{ > - u64 aligned_offset = round_down(unaligned_offset, min_block_size); > - > - return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags, > - NULL, blocks); > -} > - > static int __alloc_contig_try_harder(struct gpu_buddy *mm, > + u64 range_start, u64 range_end, > u64 size, > u64 min_block_size, > unsigned long flags, > struct list_head *blocks) > { > - u64 rhs_offset, lhs_offset, filled; > + u64 rhs_offset, lhs_offset, filled, aligned; > struct gpu_buddy_block *block; > struct rb_root *root; > struct rb_node *iter; > @@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct gpu_buddy *mm, > flags, &filled, blocks); > if (err && err != -ENOSPC) > return err; > - if (!err && IS_ALIGNED(rhs_offset, min_block_size)) > + if (!err && IS_ALIGNED(rhs_offset, min_block_size) && > + rhs_offset >= range_start && rhs_offset + size <= range_end) > return 0; > if (!err) { > /* Allocate the unaligned RHS offset using round_down */ > gpu_buddy_free_list_internal(mm, blocks); > - err = __alloc_contig_aligned_retry(mm, rhs_offset, > - size, > - min_block_size, > - flags, blocks); > - if (!err) > - return 0; > - if (err != -ENOSPC) { > - gpu_buddy_free_list_internal(mm, blocks); > - return err; > + > + aligned = round_down(rhs_offset, min_block_size); > + if (aligned >= range_start && > + aligned + size <= range_end) { > + err = __gpu_buddy_alloc_range(mm, aligned, size, > + flags, NULL, blocks); > + if (!err) > + return 0; > + if (err != -ENOSPC) { > + gpu_buddy_free_list_internal(mm, blocks); > + return err; > + } > } > goto next; > } > @@ -1759,15 +1751,17 @@ static int __alloc_contig_try_harder(struct gpu_buddy *mm, > > /* Allocate the unaligned LHS offset using round_down */ > gpu_buddy_free_list_internal(mm, blocks); > - err = __alloc_contig_aligned_retry(mm, lhs_offset, > - size, > - min_block_size, > - flags, blocks); > - if (!err) > - return 0; > - if (err != -ENOSPC) { > - gpu_buddy_free_list_internal(mm, blocks); > - return err; > + > + aligned = round_down(lhs_offset, min_block_size); > + if (aligned >= range_start && aligned + size <= range_end) { > + err = __gpu_buddy_alloc_range(mm, aligned, size, > + flags, NULL, blocks); > + if (!err) > + return 0; > + if (err != -ENOSPC) { > + gpu_buddy_free_list_internal(mm, blocks); > + return err; > + } > } > next: > gpu_buddy_free_list_internal(mm, blocks); > @@ -2021,11 +2015,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm, > min_order = ilog2(min_block_size) - ilog2(mm->chunk_size); > > if (order > mm->max_order || size > mm->size) { > - if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) && > - !(flags & GPU_BUDDY_RANGE_ALLOCATION)) > - return __alloc_contig_try_harder(mm, original_size, > + if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) { > + u64 range_start, range_end; > + > + range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? start : 0; > + range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : mm->size; > + > + return __alloc_contig_try_harder(mm, range_start, > + range_end, > + original_size, > original_min_size, > flags, blocks); > + } > > return -EINVAL; > } > @@ -2058,9 +2059,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm, > * Try contiguous block allocation through > * try harder method. > */ > - if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION && > - !(flags & GPU_BUDDY_RANGE_ALLOCATION)) { > - err = __alloc_contig_try_harder(mm, > + if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) { > + u64 range_start, range_end; > + > + range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? start : 0; > + range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : mm->size; > + > + err = __alloc_contig_try_harder(mm, range_start, > + range_end, > original_size, > original_min_size, > flags, > @@ -2068,9 +2074,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm, > if (!err) > return 0; > if (err != -ENOSPC) > - return err; > - goto err_free; > + goto err_free; > } > + > err = -ENOSPC; > goto err_free; > } while (1); > diff --git a/drivers/gpu/tests/gpu_buddy_test.c b/drivers/gpu/tests/gpu_buddy_test.c > index b75d32ca6ca0..2c445870b808 100644 > --- a/drivers/gpu/tests/gpu_buddy_test.c > +++ b/drivers/gpu/tests/gpu_buddy_test.c > @@ -1251,6 +1251,77 @@ static void gpu_test_buddy_alloc_contiguous(struct kunit *test) > gpu_buddy_fini(&mm); > } > > +static void gpu_test_buddy_alloc_range_contiguous(struct kunit *test) > +{ > + const unsigned long ps = SZ_4K, mm_size = 16 * ps; > + const unsigned long range_end = 8 * ps; > + struct gpu_buddy_block *block, *prev; > + LIST_HEAD(allocated); > + struct gpu_buddy mm; > + LIST_HEAD(pin_lo); > + LIST_HEAD(pin_hi); > + u64 total; > + > + KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps), > + "buddy_init failed\n"); > + > + /* > + * Idea is to confine the test to the sub-range [0, 32K), which a 12K > + * contiguous request (rounded up to 16K) splits into two naturally > + * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the first 4K page > + * of each slot ([0, 4K) and [16K, 20K)) so that neither aligned slot > + * can satisfy the rounded-up 16K allocation, yet the freed remainder > + * still leaves a contiguous 12K hole at offset 4K, which is page-aligned > + * but not 16K-aligned. A 12K contiguous+range allocation must therefore > + * fall back to stitching that span instead of returning -ENOSPC. > + */ > + KUNIT_ASSERT_FALSE_MSG(test, > + gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps, > + &pin_lo, 0), > + "failed to pin low page\n"); > + KUNIT_ASSERT_FALSE_MSG(test, > + gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps, > + ps, &pin_hi, 0), > + "failed to pin high page\n"); > + > + /* No aligned 16K block is free; the range-aware fallback must stitch > + * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC. > + */ > + KUNIT_ASSERT_FALSE_MSG(test, > + gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps, > + ps, &allocated, > + GPU_BUDDY_CONTIGUOUS_ALLOCATION | > + GPU_BUDDY_RANGE_ALLOCATION), > + "range-restricted contiguous alloc failed\n"); > + > + /* The result must be exactly 3*ps, contiguous, and inside the range. */ > + total = 0; > + prev = NULL; > + list_for_each_entry(block, &allocated, link) { > + u64 offset = gpu_buddy_block_offset(block); > + u64 bsize = gpu_buddy_block_size(&mm, block); > + > + KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end, > + "block [%llx, %llx) outside range\n", > + offset, offset + bsize); > + if (prev) > + KUNIT_EXPECT_EQ_MSG(test, > + gpu_buddy_block_offset(prev) + > + gpu_buddy_block_size(&mm, prev), > + offset, > + "block at %llx not contiguous\n", > + offset); > + prev = block; > + total += bsize; > + } > + KUNIT_EXPECT_EQ(test, total, 3 * ps); > + > + gpu_buddy_free_list(&mm, &allocated, 0); > + gpu_buddy_free_list(&mm, &pin_lo, 0); > + gpu_buddy_free_list(&mm, &pin_hi, 0); > + gpu_buddy_fini(&mm); > +} > + > static void gpu_test_buddy_alloc_pathological(struct kunit *test) > { > u64 mm_size, size, start = 0; > @@ -1534,10 +1605,13 @@ static void gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test) > GPU_BUDDY_RANGE_ALLOCATION); > KUNIT_EXPECT_EQ(test, err, -EINVAL); > > - /* CONTIGUOUS + RANGE should return -EINVAL (no try_harder for RANGE) */ > - err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, &blocks, > - GPU_BUDDY_CONTIGUOUS_ALLOCATION | GPU_BUDDY_RANGE_ALLOCATION); > - KUNIT_EXPECT_EQ(test, err, -EINVAL); > + /* CONTIGUOUS + RANGE should succeed via the range-aware try_harder */ > + KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, > + SZ_4K, &blocks, > + GPU_BUDDY_CONTIGUOUS_ALLOCATION | > + GPU_BUDDY_RANGE_ALLOCATION), > + "range contiguous alloc hit an error size=%llu\n", size); > + gpu_buddy_free_list(&mm, &blocks, 0); > > gpu_buddy_fini(&mm); > } > @@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = { > KUNIT_CASE(gpu_test_buddy_alloc_pessimistic), > KUNIT_CASE(gpu_test_buddy_alloc_pathological), > KUNIT_CASE(gpu_test_buddy_alloc_contiguous), > + KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous), > KUNIT_CASE(gpu_test_buddy_alloc_clear), > KUNIT_CASE(gpu_test_buddy_alloc_range), > KUNIT_CASE(gpu_test_buddy_alloc_range_bias), > > base-commit: 744f262401cc2e1f3827496c72a47e072d31a852 ^ permalink raw reply [flat|nested] 8+ messages in thread
* Re: [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback 2026-09-30 6:21 [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback Arunpravin Paneer Selvam 2026-09-30 6:31 ` sashiko-bot 2026-10-01 7:06 ` Arunpravin Paneer Selvam @ 2026-10-01 18:28 ` Matthew Auld 2026-10-05 14:14 ` Arunpravin Paneer Selvam 2 siblings, 1 reply; 8+ messages in thread From: Matthew Auld @ 2026-10-01 18:28 UTC (permalink / raw) To: Arunpravin Paneer Selvam, dri-devel, intel-gfx, intel-xe, amd-gfx Cc: christian.koenig, alexander.deucher, Anand.Raghavendra On 30/09/2026 07:21, Arunpravin Paneer Selvam wrote: > From: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com> > > A range + contiguous allocation (e.g. a scanout FB confined to the > CPU-visible VRAM aperture) rounds its size up to a power of two and > requires a naturally aligned free block of that size; on a fragmented > aperture no such aligned block may exist even though enough contiguous > space is free, so the allocation fails with -ENOSPC. > > The non-range contiguous path already recovers from this via > __alloc_contig_try_harder(), which stitches an exact-size span from > smaller adjacent blocks, but that fallback was unreachable once > GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder() > a [range_start, range_end) window and route the range + contiguous > case through it: each candidate placement is confined to the window > and aligned to min_block_size. Non-range callers pass [0, mm->size), > where the guards are no-ops, so the existing behaviour is unchanged. > A KUnit regression test covers the fragmentation pattern. > > Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b > regression. > > v2: > - Drop the split-undo patch; the range-bias search always descends to > an exact-order block or fails the split (already handled), so the > extra undo was redundant. (Matthew) > - Verified this fallback alone fixes the kms_plane regression. > > Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with decoupled dirty tracker") Patch looks more like totally new functionally/feature, so the fixes here is maybe unexpected? Did something in that fixes commit change something such that the try_harder is now needed, but that needs some expansion with bias + contig? Do we know exactly what changed here? > Assisted-by: Claude:claude-opus-4-8 Assisted-by: LLM > Cc: Matthew Auld <matthew.auld@intel.com> > Cc: Christian König <christian.koenig@amd.com> > Signed-off-by: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com> > --- > drivers/gpu/buddy.c | 88 ++++++++++++++++-------------- > drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++-- > 2 files changed, 126 insertions(+), 45 deletions(-) > > diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c > index 2f2aaadafe35..5265e1f6a318 100644 > --- a/drivers/gpu/buddy.c > +++ b/drivers/gpu/buddy.c > @@ -1685,26 +1685,14 @@ static int __gpu_buddy_alloc_range(struct gpu_buddy *mm, > blocks, total_allocated_on_err); > } > > -static int __alloc_contig_aligned_retry(struct gpu_buddy *mm, > - u64 unaligned_offset, > - u64 size, > - u64 min_block_size, > - unsigned long flags, > - struct list_head *blocks) > -{ > - u64 aligned_offset = round_down(unaligned_offset, min_block_size); > - > - return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags, > - NULL, blocks); > -} > - > static int __alloc_contig_try_harder(struct gpu_buddy *mm, > + u64 range_start, u64 range_end, > u64 size, > u64 min_block_size, > unsigned long flags, > struct list_head *blocks) > { > - u64 rhs_offset, lhs_offset, filled; > + u64 rhs_offset, lhs_offset, filled, aligned; > struct gpu_buddy_block *block; > struct rb_root *root; > struct rb_node *iter; > @@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct gpu_buddy *mm, > flags, &filled, blocks); > if (err && err != -ENOSPC) > return err; > - if (!err && IS_ALIGNED(rhs_offset, min_block_size)) > + if (!err && IS_ALIGNED(rhs_offset, min_block_size) && > + rhs_offset >= range_start && rhs_offset + size <= range_end) > return 0; > if (!err) { > /* Allocate the unaligned RHS offset using round_down */ > gpu_buddy_free_list_internal(mm, blocks); > - err = __alloc_contig_aligned_retry(mm, rhs_offset, > - size, > - min_block_size, > - flags, blocks); > - if (!err) > - return 0; > - if (err != -ENOSPC) { > - gpu_buddy_free_list_internal(mm, blocks); > - return err; > + > + aligned = round_down(rhs_offset, min_block_size); > + if (aligned >= range_start && > + aligned + size <= range_end) { > + err = __gpu_buddy_alloc_range(mm, aligned, size, > + flags, NULL, blocks); Did you consider doing a bias-range for [start, end] using min_block_size and using what it returns as the starting point, extending left/right? If that fails advance start and try again? Maybe what you have here is much better for the case you have in mind? > + if (!err) > + return 0; > + if (err != -ENOSPC) { > + gpu_buddy_free_list_internal(mm, blocks); > + return err; > + } > } > goto next; > } > @@ -1759,15 +1751,17 @@ static int __alloc_contig_try_harder(struct gpu_buddy *mm, > > /* Allocate the unaligned LHS offset using round_down */ > gpu_buddy_free_list_internal(mm, blocks); > - err = __alloc_contig_aligned_retry(mm, lhs_offset, > - size, > - min_block_size, > - flags, blocks); > - if (!err) > - return 0; > - if (err != -ENOSPC) { > - gpu_buddy_free_list_internal(mm, blocks); > - return err; > + > + aligned = round_down(lhs_offset, min_block_size); > + if (aligned >= range_start && aligned + size <= range_end) { > + err = __gpu_buddy_alloc_range(mm, aligned, size, > + flags, NULL, blocks); > + if (!err) > + return 0; > + if (err != -ENOSPC) { > + gpu_buddy_free_list_internal(mm, blocks); > + return err; > + } > } > next: > gpu_buddy_free_list_internal(mm, blocks); > @@ -2021,11 +2015,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm, > min_order = ilog2(min_block_size) - ilog2(mm->chunk_size); > > if (order > mm->max_order || size > mm->size) { > - if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) && > - !(flags & GPU_BUDDY_RANGE_ALLOCATION)) > - return __alloc_contig_try_harder(mm, original_size, > + if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) { > + u64 range_start, range_end; > + > + range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? start : 0; > + range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : mm->size; > + > + return __alloc_contig_try_harder(mm, range_start, > + range_end, > + original_size, > original_min_size, > flags, blocks); > + } > > return -EINVAL; > } > @@ -2058,9 +2059,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm, > * Try contiguous block allocation through > * try harder method. > */ > - if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION && > - !(flags & GPU_BUDDY_RANGE_ALLOCATION)) { > - err = __alloc_contig_try_harder(mm, > + if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) { > + u64 range_start, range_end; > + > + range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? start : 0; > + range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : mm->size; > + > + err = __alloc_contig_try_harder(mm, range_start, > + range_end, > original_size, > original_min_size, > flags, > @@ -2068,9 +2074,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm, > if (!err) > return 0; > if (err != -ENOSPC) > - return err; > - goto err_free; > + goto err_free; > } > + > err = -ENOSPC; > goto err_free; > } while (1); > diff --git a/drivers/gpu/tests/gpu_buddy_test.c b/drivers/gpu/tests/gpu_buddy_test.c > index b75d32ca6ca0..2c445870b808 100644 > --- a/drivers/gpu/tests/gpu_buddy_test.c > +++ b/drivers/gpu/tests/gpu_buddy_test.c > @@ -1251,6 +1251,77 @@ static void gpu_test_buddy_alloc_contiguous(struct kunit *test) > gpu_buddy_fini(&mm); > } > > +static void gpu_test_buddy_alloc_range_contiguous(struct kunit *test) > +{ > + const unsigned long ps = SZ_4K, mm_size = 16 * ps; > + const unsigned long range_end = 8 * ps; > + struct gpu_buddy_block *block, *prev; > + LIST_HEAD(allocated); > + struct gpu_buddy mm; > + LIST_HEAD(pin_lo); > + LIST_HEAD(pin_hi); > + u64 total; > + > + KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps), > + "buddy_init failed\n"); > + > + /* > + * Idea is to confine the test to the sub-range [0, 32K), which a 12K > + * contiguous request (rounded up to 16K) splits into two naturally > + * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the first 4K page > + * of each slot ([0, 4K) and [16K, 20K)) so that neither aligned slot > + * can satisfy the rounded-up 16K allocation, yet the freed remainder > + * still leaves a contiguous 12K hole at offset 4K, which is page-aligned > + * but not 16K-aligned. A 12K contiguous+range allocation must therefore > + * fall back to stitching that span instead of returning -ENOSPC. > + */ > + KUNIT_ASSERT_FALSE_MSG(test, > + gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps, > + &pin_lo, 0), > + "failed to pin low page\n"); > + KUNIT_ASSERT_FALSE_MSG(test, > + gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps, > + ps, &pin_hi, 0), > + "failed to pin high page\n"); > + > + /* No aligned 16K block is free; the range-aware fallback must stitch > + * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC. > + */ > + KUNIT_ASSERT_FALSE_MSG(test, > + gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps, > + ps, &allocated, > + GPU_BUDDY_CONTIGUOUS_ALLOCATION | > + GPU_BUDDY_RANGE_ALLOCATION), > + "range-restricted contiguous alloc failed\n"); > + > + /* The result must be exactly 3*ps, contiguous, and inside the range. */ > + total = 0; > + prev = NULL; > + list_for_each_entry(block, &allocated, link) { > + u64 offset = gpu_buddy_block_offset(block); > + u64 bsize = gpu_buddy_block_size(&mm, block); > + > + KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end, > + "block [%llx, %llx) outside range\n", > + offset, offset + bsize); > + if (prev) > + KUNIT_EXPECT_EQ_MSG(test, > + gpu_buddy_block_offset(prev) + > + gpu_buddy_block_size(&mm, prev), > + offset, > + "block at %llx not contiguous\n", > + offset); > + prev = block; > + total += bsize; > + } > + KUNIT_EXPECT_EQ(test, total, 3 * ps); > + > + gpu_buddy_free_list(&mm, &allocated, 0); > + gpu_buddy_free_list(&mm, &pin_lo, 0); > + gpu_buddy_free_list(&mm, &pin_hi, 0); > + gpu_buddy_fini(&mm); > +} > + > static void gpu_test_buddy_alloc_pathological(struct kunit *test) > { > u64 mm_size, size, start = 0; > @@ -1534,10 +1605,13 @@ static void gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test) > GPU_BUDDY_RANGE_ALLOCATION); > KUNIT_EXPECT_EQ(test, err, -EINVAL); > > - /* CONTIGUOUS + RANGE should return -EINVAL (no try_harder for RANGE) */ > - err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, &blocks, > - GPU_BUDDY_CONTIGUOUS_ALLOCATION | GPU_BUDDY_RANGE_ALLOCATION); > - KUNIT_EXPECT_EQ(test, err, -EINVAL); > + /* CONTIGUOUS + RANGE should succeed via the range-aware try_harder */ > + KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, > + SZ_4K, &blocks, > + GPU_BUDDY_CONTIGUOUS_ALLOCATION | > + GPU_BUDDY_RANGE_ALLOCATION), > + "range contiguous alloc hit an error size=%llu\n", size); > + gpu_buddy_free_list(&mm, &blocks, 0); > > gpu_buddy_fini(&mm); > } > @@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = { > KUNIT_CASE(gpu_test_buddy_alloc_pessimistic), > KUNIT_CASE(gpu_test_buddy_alloc_pathological), > KUNIT_CASE(gpu_test_buddy_alloc_contiguous), > + KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous), > KUNIT_CASE(gpu_test_buddy_alloc_clear), > KUNIT_CASE(gpu_test_buddy_alloc_range), > KUNIT_CASE(gpu_test_buddy_alloc_range_bias), > > base-commit: 744f262401cc2e1f3827496c72a47e072d31a852 ^ permalink raw reply [flat|nested] 8+ messages in thread
* Re: [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback 2026-10-01 18:28 ` Matthew Auld @ 2026-10-05 14:14 ` Arunpravin Paneer Selvam 2026-10-06 9:57 ` Matthew Auld 0 siblings, 1 reply; 8+ messages in thread From: Arunpravin Paneer Selvam @ 2026-10-05 14:14 UTC (permalink / raw) To: Matthew Auld, dri-devel, intel-gfx, intel-xe, amd-gfx Cc: christian.koenig, alexander.deucher, Anand.Raghavendra On 10/1/2026 11:58 PM, Matthew Auld wrote: > On 30/09/2026 07:21, Arunpravin Paneer Selvam wrote: >> From: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com> >> >> A range + contiguous allocation (e.g. a scanout FB confined to the >> CPU-visible VRAM aperture) rounds its size up to a power of two and >> requires a naturally aligned free block of that size; on a fragmented >> aperture no such aligned block may exist even though enough contiguous >> space is free, so the allocation fails with -ENOSPC. >> >> The non-range contiguous path already recovers from this via >> __alloc_contig_try_harder(), which stitches an exact-size span from >> smaller adjacent blocks, but that fallback was unreachable once >> GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder() >> a [range_start, range_end) window and route the range + contiguous >> case through it: each candidate placement is confined to the window >> and aligned to min_block_size. Non-range callers pass [0, mm->size), >> where the guards are no-ops, so the existing behaviour is unchanged. >> A KUnit regression test covers the fragmentation pattern. >> >> Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b >> regression. >> >> v2: >> - Drop the split-undo patch; the range-bias search always descends to >> an exact-order block or fails the split (already handled), so the >> extra undo was redundant. (Matthew) >> - Verified this fallback alone fixes the kms_plane regression. >> >> Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with >> decoupled dirty tracker") > > Patch looks more like totally new functionally/feature, so the fixes > here is maybe unexpected? Did something in that fixes commit change > something such that the try_harder is now needed, but that needs some > expansion with bias + contig? Do we know exactly what changed here? That commit removed __force_merge(), which previously helped recover larger contiguous allocations on demand. As a result, some range-restricted contiguous allocation requests that used to succeed now fail with -ENOSPC (the kms_plane regression). This patch restores that capability through the try_harder() stitching logic, so the Fixes: tag reflects a regression introduced by that commit rather than new functionality. > >> Assisted-by: Claude:claude-opus-4-8 > > Assisted-by: LLM > >> Cc: Matthew Auld <matthew.auld@intel.com> >> Cc: Christian König <christian.koenig@amd.com> >> Signed-off-by: Arunpravin Paneer Selvam >> <Arunpravin.PaneerSelvam@amd.com> >> --- >> drivers/gpu/buddy.c | 88 ++++++++++++++++-------------- >> drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++-- >> 2 files changed, 126 insertions(+), 45 deletions(-) >> >> diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c >> index 2f2aaadafe35..5265e1f6a318 100644 >> --- a/drivers/gpu/buddy.c >> +++ b/drivers/gpu/buddy.c >> @@ -1685,26 +1685,14 @@ static int __gpu_buddy_alloc_range(struct >> gpu_buddy *mm, >> blocks, total_allocated_on_err); >> } >> -static int __alloc_contig_aligned_retry(struct gpu_buddy *mm, >> - u64 unaligned_offset, >> - u64 size, >> - u64 min_block_size, >> - unsigned long flags, >> - struct list_head *blocks) >> -{ >> - u64 aligned_offset = round_down(unaligned_offset, min_block_size); >> - >> - return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags, >> - NULL, blocks); >> -} >> - >> static int __alloc_contig_try_harder(struct gpu_buddy *mm, >> + u64 range_start, u64 range_end, >> u64 size, >> u64 min_block_size, >> unsigned long flags, >> struct list_head *blocks) >> { >> - u64 rhs_offset, lhs_offset, filled; >> + u64 rhs_offset, lhs_offset, filled, aligned; >> struct gpu_buddy_block *block; >> struct rb_root *root; >> struct rb_node *iter; >> @@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct >> gpu_buddy *mm, >> flags, &filled, blocks); >> if (err && err != -ENOSPC) >> return err; >> - if (!err && IS_ALIGNED(rhs_offset, min_block_size)) >> + if (!err && IS_ALIGNED(rhs_offset, min_block_size) && >> + rhs_offset >= range_start && rhs_offset + size <= >> range_end) >> return 0; >> if (!err) { >> /* Allocate the unaligned RHS offset using round_down */ >> gpu_buddy_free_list_internal(mm, blocks); >> - err = __alloc_contig_aligned_retry(mm, rhs_offset, >> - size, >> - min_block_size, >> - flags, blocks); >> - if (!err) >> - return 0; >> - if (err != -ENOSPC) { >> - gpu_buddy_free_list_internal(mm, blocks); >> - return err; >> + >> + aligned = round_down(rhs_offset, min_block_size); >> + if (aligned >= range_start && >> + aligned + size <= range_end) { >> + err = __gpu_buddy_alloc_range(mm, aligned, size, >> + flags, NULL, blocks); > > Did you consider doing a bias-range for [start, end] using > min_block_size and using what it returns as the starting point, > extending left/right? If that fails advance start and try again? Maybe > what you have here is much better for the case you have in mind? I took a closer look at the alloc_range_bias approach. My understanding is that this would repeatedly bias within [start, end], use the returned min_block_size block as a starting point, and then extend left/right to build the requested run. While that should work, every failed attempt may require splitting higher-order blocks to obtain a min_block_size block, followed by a free/merge back when the extension fails. The free-tree descent evaluates the available starting points through a tree walk without any splitting or allocation just for enumeration, so I kept that approach. Please let me know if there is a case where alloc_range_bias would provide an advantage over the free-tree descent. While evaluating it, I found two issues in the current try_harder() implementation: 1. It only searches free_tree[size_order]. This is insufficient when min_block_size < size. For example, for a 128 KiB allocation (order-5) with min_block_size = 4 KiB, free_tree[5] may be empty while two adjacent order-4 (64 KiB) blocks covering [64 KiB, 128 KiB] and [128 KiB, 192 KiB] are free. These blocks still form a valid order-5-sized contiguous run, but they can never appear in free_tree[5] because they are not mergeable buddies. As a result, a search that only considers free_tree[5] will never find this placement. v3 addresses this by walking the free tree from the requested size order down to min_order, while reusing the existing extend/stitch logic and min_block_size alignment checks unchanged. 2. The search is not restricted to the requested range. v3 confines the walk to [range_start, range_end], skipping candidates at or above range_end and stopping once the walk falls below range_start. This ensures that range-constrained allocations only consider free blocks within the requested window and avoids performing allocation/free attempts on blocks outside the specified range. For non-range callers, the effective window remains [0, mm->size], so the behavior is unchanged. Regards, Arun. > >> + if (!err) >> + return 0; >> + if (err != -ENOSPC) { >> + gpu_buddy_free_list_internal(mm, blocks); >> + return err; >> + } >> } >> goto next; >> } >> @@ -1759,15 +1751,17 @@ static int __alloc_contig_try_harder(struct >> gpu_buddy *mm, >> /* Allocate the unaligned LHS offset using round_down */ >> gpu_buddy_free_list_internal(mm, blocks); >> - err = __alloc_contig_aligned_retry(mm, lhs_offset, >> - size, >> - min_block_size, >> - flags, blocks); >> - if (!err) >> - return 0; >> - if (err != -ENOSPC) { >> - gpu_buddy_free_list_internal(mm, blocks); >> - return err; >> + >> + aligned = round_down(lhs_offset, min_block_size); >> + if (aligned >= range_start && aligned + size <= range_end) { >> + err = __gpu_buddy_alloc_range(mm, aligned, size, >> + flags, NULL, blocks); >> + if (!err) >> + return 0; >> + if (err != -ENOSPC) { >> + gpu_buddy_free_list_internal(mm, blocks); >> + return err; >> + } >> } >> next: >> gpu_buddy_free_list_internal(mm, blocks); >> @@ -2021,11 +2015,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm, >> min_order = ilog2(min_block_size) - ilog2(mm->chunk_size); >> if (order > mm->max_order || size > mm->size) { >> - if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) && >> - !(flags & GPU_BUDDY_RANGE_ALLOCATION)) >> - return __alloc_contig_try_harder(mm, original_size, >> + if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) { >> + u64 range_start, range_end; >> + >> + range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? >> start : 0; >> + range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : >> mm->size; >> + >> + return __alloc_contig_try_harder(mm, range_start, >> + range_end, >> + original_size, >> original_min_size, >> flags, blocks); >> + } >> return -EINVAL; >> } >> @@ -2058,9 +2059,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm, >> * Try contiguous block allocation through >> * try harder method. >> */ >> - if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION && >> - !(flags & GPU_BUDDY_RANGE_ALLOCATION)) { >> - err = __alloc_contig_try_harder(mm, >> + if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) { >> + u64 range_start, range_end; >> + >> + range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? >> start : 0; >> + range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? >> end : mm->size; >> + >> + err = __alloc_contig_try_harder(mm, range_start, >> + range_end, >> original_size, >> original_min_size, >> flags, >> @@ -2068,9 +2074,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm, >> if (!err) >> return 0; >> if (err != -ENOSPC) >> - return err; >> - goto err_free; >> + goto err_free; >> } >> + >> err = -ENOSPC; >> goto err_free; >> } while (1); >> diff --git a/drivers/gpu/tests/gpu_buddy_test.c >> b/drivers/gpu/tests/gpu_buddy_test.c >> index b75d32ca6ca0..2c445870b808 100644 >> --- a/drivers/gpu/tests/gpu_buddy_test.c >> +++ b/drivers/gpu/tests/gpu_buddy_test.c >> @@ -1251,6 +1251,77 @@ static void >> gpu_test_buddy_alloc_contiguous(struct kunit *test) >> gpu_buddy_fini(&mm); >> } >> +static void gpu_test_buddy_alloc_range_contiguous(struct kunit *test) >> +{ >> + const unsigned long ps = SZ_4K, mm_size = 16 * ps; >> + const unsigned long range_end = 8 * ps; >> + struct gpu_buddy_block *block, *prev; >> + LIST_HEAD(allocated); >> + struct gpu_buddy mm; >> + LIST_HEAD(pin_lo); >> + LIST_HEAD(pin_hi); >> + u64 total; >> + >> + KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps), >> + "buddy_init failed\n"); >> + >> + /* >> + * Idea is to confine the test to the sub-range [0, 32K), which >> a 12K >> + * contiguous request (rounded up to 16K) splits into two naturally >> + * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the first >> 4K page >> + * of each slot ([0, 4K) and [16K, 20K)) so that neither aligned >> slot >> + * can satisfy the rounded-up 16K allocation, yet the freed >> remainder >> + * still leaves a contiguous 12K hole at offset 4K, which is >> page-aligned >> + * but not 16K-aligned. A 12K contiguous+range allocation must >> therefore >> + * fall back to stitching that span instead of returning -ENOSPC. >> + */ >> + KUNIT_ASSERT_FALSE_MSG(test, >> + gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps, >> + &pin_lo, 0), >> + "failed to pin low page\n"); >> + KUNIT_ASSERT_FALSE_MSG(test, >> + gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps, >> + ps, &pin_hi, 0), >> + "failed to pin high page\n"); >> + >> + /* No aligned 16K block is free; the range-aware fallback must >> stitch >> + * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC. >> + */ >> + KUNIT_ASSERT_FALSE_MSG(test, >> + gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps, >> + ps, &allocated, >> + GPU_BUDDY_CONTIGUOUS_ALLOCATION | >> + GPU_BUDDY_RANGE_ALLOCATION), >> + "range-restricted contiguous alloc failed\n"); >> + >> + /* The result must be exactly 3*ps, contiguous, and inside the >> range. */ >> + total = 0; >> + prev = NULL; >> + list_for_each_entry(block, &allocated, link) { >> + u64 offset = gpu_buddy_block_offset(block); >> + u64 bsize = gpu_buddy_block_size(&mm, block); >> + >> + KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end, >> + "block [%llx, %llx) outside range\n", >> + offset, offset + bsize); >> + if (prev) >> + KUNIT_EXPECT_EQ_MSG(test, >> + gpu_buddy_block_offset(prev) + >> + gpu_buddy_block_size(&mm, prev), >> + offset, >> + "block at %llx not contiguous\n", >> + offset); >> + prev = block; >> + total += bsize; >> + } >> + KUNIT_EXPECT_EQ(test, total, 3 * ps); >> + >> + gpu_buddy_free_list(&mm, &allocated, 0); >> + gpu_buddy_free_list(&mm, &pin_lo, 0); >> + gpu_buddy_free_list(&mm, &pin_hi, 0); >> + gpu_buddy_fini(&mm); >> +} >> + >> static void gpu_test_buddy_alloc_pathological(struct kunit *test) >> { >> u64 mm_size, size, start = 0; >> @@ -1534,10 +1605,13 @@ static void >> gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test) >> GPU_BUDDY_RANGE_ALLOCATION); >> KUNIT_EXPECT_EQ(test, err, -EINVAL); >> - /* CONTIGUOUS + RANGE should return -EINVAL (no try_harder for >> RANGE) */ >> - err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, &blocks, >> - GPU_BUDDY_CONTIGUOUS_ALLOCATION | >> GPU_BUDDY_RANGE_ALLOCATION); >> - KUNIT_EXPECT_EQ(test, err, -EINVAL); >> + /* CONTIGUOUS + RANGE should succeed via the range-aware >> try_harder */ >> + KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, >> mm_size, size, >> + SZ_4K, &blocks, >> + GPU_BUDDY_CONTIGUOUS_ALLOCATION | >> + GPU_BUDDY_RANGE_ALLOCATION), >> + "range contiguous alloc hit an error >> size=%llu\n", size); >> + gpu_buddy_free_list(&mm, &blocks, 0); >> gpu_buddy_fini(&mm); >> } >> @@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = { >> KUNIT_CASE(gpu_test_buddy_alloc_pessimistic), >> KUNIT_CASE(gpu_test_buddy_alloc_pathological), >> KUNIT_CASE(gpu_test_buddy_alloc_contiguous), >> + KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous), >> KUNIT_CASE(gpu_test_buddy_alloc_clear), >> KUNIT_CASE(gpu_test_buddy_alloc_range), >> KUNIT_CASE(gpu_test_buddy_alloc_range_bias), >> >> base-commit: 744f262401cc2e1f3827496c72a47e072d31a852 > ^ permalink raw reply [flat|nested] 8+ messages in thread
* Re: [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback 2026-10-05 14:14 ` Arunpravin Paneer Selvam @ 2026-10-06 9:57 ` Matthew Auld 2026-10-09 9:14 ` Arunpravin Paneer Selvam 0 siblings, 1 reply; 8+ messages in thread From: Matthew Auld @ 2026-10-06 9:57 UTC (permalink / raw) To: Arunpravin Paneer Selvam, dri-devel, intel-gfx, intel-xe, amd-gfx Cc: christian.koenig, alexander.deucher, Anand.Raghavendra On 05/10/2026 15:14, Arunpravin Paneer Selvam wrote: > > > On 10/1/2026 11:58 PM, Matthew Auld wrote: >> On 30/09/2026 07:21, Arunpravin Paneer Selvam wrote: >>> From: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com> >>> >>> A range + contiguous allocation (e.g. a scanout FB confined to the >>> CPU-visible VRAM aperture) rounds its size up to a power of two and >>> requires a naturally aligned free block of that size; on a fragmented >>> aperture no such aligned block may exist even though enough contiguous >>> space is free, so the allocation fails with -ENOSPC. >>> >>> The non-range contiguous path already recovers from this via >>> __alloc_contig_try_harder(), which stitches an exact-size span from >>> smaller adjacent blocks, but that fallback was unreachable once >>> GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder() >>> a [range_start, range_end) window and route the range + contiguous >>> case through it: each candidate placement is confined to the window >>> and aligned to min_block_size. Non-range callers pass [0, mm->size), >>> where the guards are no-ops, so the existing behaviour is unchanged. >>> A KUnit regression test covers the fragmentation pattern. >>> >>> Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b >>> regression. >>> >>> v2: >>> - Drop the split-undo patch; the range-bias search always descends to >>> an exact-order block or fails the split (already handled), so the >>> extra undo was redundant. (Matthew) >>> - Verified this fallback alone fixes the kms_plane regression. >>> >>> Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with >>> decoupled dirty tracker") >> >> Patch looks more like totally new functionally/feature, so the fixes >> here is maybe unexpected? Did something in that fixes commit change >> something such that the try_harder is now needed, but that needs some >> expansion with bias + contig? Do we know exactly what changed here? > That commit removed __force_merge(), which previously helped recover > larger contiguous allocations on demand. As a result, some range- __force_merge() got nuked, but IIRC I think that was essentially because we now "force merge" on free, so shouldn't we get the ~same result? Or is the fact that we force_merge() on every free giving a different layout of pages, and with some very some specific allocation pattern that difference subtly results in -ENOSPC, somehow? > restricted contiguous allocation requests > that used to succeed now fail with -ENOSPC (the kms_plane regression). > This patch restores that capability through the try_harder() stitching > logic, so the Fixes: tag reflects a > regression introduced by that commit rather than new functionality. >> >>> Assisted-by: Claude:claude-opus-4-8 >> >> Assisted-by: LLM >> >>> Cc: Matthew Auld <matthew.auld@intel.com> >>> Cc: Christian König <christian.koenig@amd.com> >>> Signed-off-by: Arunpravin Paneer Selvam >>> <Arunpravin.PaneerSelvam@amd.com> >>> --- >>> drivers/gpu/buddy.c | 88 ++++++++++++++++-------------- >>> drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++-- >>> 2 files changed, 126 insertions(+), 45 deletions(-) >>> >>> diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c >>> index 2f2aaadafe35..5265e1f6a318 100644 >>> --- a/drivers/gpu/buddy.c >>> +++ b/drivers/gpu/buddy.c >>> @@ -1685,26 +1685,14 @@ static int __gpu_buddy_alloc_range(struct >>> gpu_buddy *mm, >>> blocks, total_allocated_on_err); >>> } >>> -static int __alloc_contig_aligned_retry(struct gpu_buddy *mm, >>> - u64 unaligned_offset, >>> - u64 size, >>> - u64 min_block_size, >>> - unsigned long flags, >>> - struct list_head *blocks) >>> -{ >>> - u64 aligned_offset = round_down(unaligned_offset, min_block_size); >>> - >>> - return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags, >>> - NULL, blocks); >>> -} >>> - >>> static int __alloc_contig_try_harder(struct gpu_buddy *mm, >>> + u64 range_start, u64 range_end, >>> u64 size, >>> u64 min_block_size, >>> unsigned long flags, >>> struct list_head *blocks) >>> { >>> - u64 rhs_offset, lhs_offset, filled; >>> + u64 rhs_offset, lhs_offset, filled, aligned; >>> struct gpu_buddy_block *block; >>> struct rb_root *root; >>> struct rb_node *iter; >>> @@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct >>> gpu_buddy *mm, >>> flags, &filled, blocks); >>> if (err && err != -ENOSPC) >>> return err; >>> - if (!err && IS_ALIGNED(rhs_offset, min_block_size)) >>> + if (!err && IS_ALIGNED(rhs_offset, min_block_size) && >>> + rhs_offset >= range_start && rhs_offset + size <= >>> range_end) >>> return 0; >>> if (!err) { >>> /* Allocate the unaligned RHS offset using round_down */ >>> gpu_buddy_free_list_internal(mm, blocks); >>> - err = __alloc_contig_aligned_retry(mm, rhs_offset, >>> - size, >>> - min_block_size, >>> - flags, blocks); >>> - if (!err) >>> - return 0; >>> - if (err != -ENOSPC) { >>> - gpu_buddy_free_list_internal(mm, blocks); >>> - return err; >>> + >>> + aligned = round_down(rhs_offset, min_block_size); >>> + if (aligned >= range_start && >>> + aligned + size <= range_end) { >>> + err = __gpu_buddy_alloc_range(mm, aligned, size, >>> + flags, NULL, blocks); >> >> Did you consider doing a bias-range for [start, end] using >> min_block_size and using what it returns as the starting point, >> extending left/right? If that fails advance start and try again? Maybe >> what you have here is much better for the case you have in mind? > I took a closer look at the alloc_range_bias approach. My understanding > is that this would repeatedly bias within [start, end], use the returned > min_block_size block as a starting point, and then extend left/right to > build the requested run. > While that should work, every failed attempt may require splitting > higher-order blocks to obtain a min_block_size block, followed by a > free/merge back when the extension fails. > The free-tree descent evaluates the available starting points through a > tree walk without any splitting or allocation just for enumeration, so I > kept that approach. > Please let me know if there is a case where alloc_range_bias would > provide an advantage over the free-tree descent. > > While evaluating it, I found two issues in the current try_harder() > implementation: > > 1. It only searches free_tree[size_order]. This is insufficient when > min_block_size < size. For example, for a 128 KiB allocation (order-5) > with min_block_size = 4 KiB, free_tree[5] may be empty > while two adjacent order-4 (64 KiB) blocks covering [64 KiB, 128 KiB] > and [128 KiB, 192 KiB] are free. These blocks still form a valid > order-5-sized contiguous run, but they can never appear in > free_tree[5] because they are not mergeable buddies. As a result, a > search that only considers free_tree[5] will never find this placement. > v3 addresses this by walking the free tree from the > requested size order down to min_order, while reusing the existing > extend/stitch logic and min_block_size alignment checks unchanged. > > 2. The search is not restricted to the requested range. v3 confines the > walk to [range_start, range_end], skipping candidates at or above > range_end and stopping once the walk falls below range_start. > This ensures that range-constrained allocations only consider free > blocks within the requested window and avoids performing allocation/free > attempts on blocks outside the specified range. > For non-range callers, the effective window remains [0, mm->size], so > the behavior is unchanged. > > Regards, > Arun. >> >>> + if (!err) >>> + return 0; >>> + if (err != -ENOSPC) { >>> + gpu_buddy_free_list_internal(mm, blocks); >>> + return err; >>> + } >>> } >>> goto next; >>> } >>> @@ -1759,15 +1751,17 @@ static int __alloc_contig_try_harder(struct >>> gpu_buddy *mm, >>> /* Allocate the unaligned LHS offset using round_down */ >>> gpu_buddy_free_list_internal(mm, blocks); >>> - err = __alloc_contig_aligned_retry(mm, lhs_offset, >>> - size, >>> - min_block_size, >>> - flags, blocks); >>> - if (!err) >>> - return 0; >>> - if (err != -ENOSPC) { >>> - gpu_buddy_free_list_internal(mm, blocks); >>> - return err; >>> + >>> + aligned = round_down(lhs_offset, min_block_size); >>> + if (aligned >= range_start && aligned + size <= range_end) { >>> + err = __gpu_buddy_alloc_range(mm, aligned, size, >>> + flags, NULL, blocks); >>> + if (!err) >>> + return 0; >>> + if (err != -ENOSPC) { >>> + gpu_buddy_free_list_internal(mm, blocks); >>> + return err; >>> + } >>> } >>> next: >>> gpu_buddy_free_list_internal(mm, blocks); >>> @@ -2021,11 +2015,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm, >>> min_order = ilog2(min_block_size) - ilog2(mm->chunk_size); >>> if (order > mm->max_order || size > mm->size) { >>> - if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) && >>> - !(flags & GPU_BUDDY_RANGE_ALLOCATION)) >>> - return __alloc_contig_try_harder(mm, original_size, >>> + if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) { >>> + u64 range_start, range_end; >>> + >>> + range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? >>> start : 0; >>> + range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end : >>> mm->size; >>> + >>> + return __alloc_contig_try_harder(mm, range_start, >>> + range_end, >>> + original_size, >>> original_min_size, >>> flags, blocks); >>> + } >>> return -EINVAL; >>> } >>> @@ -2058,9 +2059,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm, >>> * Try contiguous block allocation through >>> * try harder method. >>> */ >>> - if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION && >>> - !(flags & GPU_BUDDY_RANGE_ALLOCATION)) { >>> - err = __alloc_contig_try_harder(mm, >>> + if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) { >>> + u64 range_start, range_end; >>> + >>> + range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? >>> start : 0; >>> + range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? >>> end : mm->size; >>> + >>> + err = __alloc_contig_try_harder(mm, range_start, >>> + range_end, >>> original_size, >>> original_min_size, >>> flags, >>> @@ -2068,9 +2074,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm, >>> if (!err) >>> return 0; >>> if (err != -ENOSPC) >>> - return err; >>> - goto err_free; >>> + goto err_free; >>> } >>> + >>> err = -ENOSPC; >>> goto err_free; >>> } while (1); >>> diff --git a/drivers/gpu/tests/gpu_buddy_test.c b/drivers/gpu/tests/ >>> gpu_buddy_test.c >>> index b75d32ca6ca0..2c445870b808 100644 >>> --- a/drivers/gpu/tests/gpu_buddy_test.c >>> +++ b/drivers/gpu/tests/gpu_buddy_test.c >>> @@ -1251,6 +1251,77 @@ static void >>> gpu_test_buddy_alloc_contiguous(struct kunit *test) >>> gpu_buddy_fini(&mm); >>> } >>> +static void gpu_test_buddy_alloc_range_contiguous(struct kunit *test) >>> +{ >>> + const unsigned long ps = SZ_4K, mm_size = 16 * ps; >>> + const unsigned long range_end = 8 * ps; >>> + struct gpu_buddy_block *block, *prev; >>> + LIST_HEAD(allocated); >>> + struct gpu_buddy mm; >>> + LIST_HEAD(pin_lo); >>> + LIST_HEAD(pin_hi); >>> + u64 total; >>> + >>> + KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps), >>> + "buddy_init failed\n"); >>> + >>> + /* >>> + * Idea is to confine the test to the sub-range [0, 32K), which >>> a 12K >>> + * contiguous request (rounded up to 16K) splits into two naturally >>> + * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the first >>> 4K page >>> + * of each slot ([0, 4K) and [16K, 20K)) so that neither aligned >>> slot >>> + * can satisfy the rounded-up 16K allocation, yet the freed >>> remainder >>> + * still leaves a contiguous 12K hole at offset 4K, which is >>> page-aligned >>> + * but not 16K-aligned. A 12K contiguous+range allocation must >>> therefore >>> + * fall back to stitching that span instead of returning -ENOSPC. >>> + */ >>> + KUNIT_ASSERT_FALSE_MSG(test, >>> + gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps, >>> + &pin_lo, 0), >>> + "failed to pin low page\n"); >>> + KUNIT_ASSERT_FALSE_MSG(test, >>> + gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps, >>> + ps, &pin_hi, 0), >>> + "failed to pin high page\n"); >>> + >>> + /* No aligned 16K block is free; the range-aware fallback must >>> stitch >>> + * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC. >>> + */ >>> + KUNIT_ASSERT_FALSE_MSG(test, >>> + gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps, >>> + ps, &allocated, >>> + GPU_BUDDY_CONTIGUOUS_ALLOCATION | >>> + GPU_BUDDY_RANGE_ALLOCATION), >>> + "range-restricted contiguous alloc failed\n"); >>> + >>> + /* The result must be exactly 3*ps, contiguous, and inside the >>> range. */ >>> + total = 0; >>> + prev = NULL; >>> + list_for_each_entry(block, &allocated, link) { >>> + u64 offset = gpu_buddy_block_offset(block); >>> + u64 bsize = gpu_buddy_block_size(&mm, block); >>> + >>> + KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end, >>> + "block [%llx, %llx) outside range\n", >>> + offset, offset + bsize); >>> + if (prev) >>> + KUNIT_EXPECT_EQ_MSG(test, >>> + gpu_buddy_block_offset(prev) + >>> + gpu_buddy_block_size(&mm, prev), >>> + offset, >>> + "block at %llx not contiguous\n", >>> + offset); >>> + prev = block; >>> + total += bsize; >>> + } >>> + KUNIT_EXPECT_EQ(test, total, 3 * ps); >>> + >>> + gpu_buddy_free_list(&mm, &allocated, 0); >>> + gpu_buddy_free_list(&mm, &pin_lo, 0); >>> + gpu_buddy_free_list(&mm, &pin_hi, 0); >>> + gpu_buddy_fini(&mm); >>> +} >>> + >>> static void gpu_test_buddy_alloc_pathological(struct kunit *test) >>> { >>> u64 mm_size, size, start = 0; >>> @@ -1534,10 +1605,13 @@ static void >>> gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test) >>> GPU_BUDDY_RANGE_ALLOCATION); >>> KUNIT_EXPECT_EQ(test, err, -EINVAL); >>> - /* CONTIGUOUS + RANGE should return -EINVAL (no try_harder for >>> RANGE) */ >>> - err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, &blocks, >>> - GPU_BUDDY_CONTIGUOUS_ALLOCATION | >>> GPU_BUDDY_RANGE_ALLOCATION); >>> - KUNIT_EXPECT_EQ(test, err, -EINVAL); >>> + /* CONTIGUOUS + RANGE should succeed via the range-aware >>> try_harder */ >>> + KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, >>> mm_size, size, >>> + SZ_4K, &blocks, >>> + GPU_BUDDY_CONTIGUOUS_ALLOCATION | >>> + GPU_BUDDY_RANGE_ALLOCATION), >>> + "range contiguous alloc hit an error >>> size=%llu\n", size); >>> + gpu_buddy_free_list(&mm, &blocks, 0); >>> gpu_buddy_fini(&mm); >>> } >>> @@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = { >>> KUNIT_CASE(gpu_test_buddy_alloc_pessimistic), >>> KUNIT_CASE(gpu_test_buddy_alloc_pathological), >>> KUNIT_CASE(gpu_test_buddy_alloc_contiguous), >>> + KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous), >>> KUNIT_CASE(gpu_test_buddy_alloc_clear), >>> KUNIT_CASE(gpu_test_buddy_alloc_range), >>> KUNIT_CASE(gpu_test_buddy_alloc_range_bias), >>> >>> base-commit: 744f262401cc2e1f3827496c72a47e072d31a852 >> > ^ permalink raw reply [flat|nested] 8+ messages in thread
* Re: [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback 2026-10-06 9:57 ` Matthew Auld @ 2026-10-09 9:14 ` Arunpravin Paneer Selvam 2026-10-09 10:17 ` Matthew Auld 0 siblings, 1 reply; 8+ messages in thread From: Arunpravin Paneer Selvam @ 2026-10-09 9:14 UTC (permalink / raw) To: Matthew Auld, dri-devel, intel-gfx, intel-xe, amd-gfx Cc: christian.koenig, alexander.deucher, Anand.Raghavendra On 10/6/2026 3:27 PM, Matthew Auld wrote: > On 05/10/2026 15:14, Arunpravin Paneer Selvam wrote: >> >> >> On 10/1/2026 11:58 PM, Matthew Auld wrote: >>> On 30/09/2026 07:21, Arunpravin Paneer Selvam wrote: >>>> From: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com> >>>> >>>> A range + contiguous allocation (e.g. a scanout FB confined to the >>>> CPU-visible VRAM aperture) rounds its size up to a power of two and >>>> requires a naturally aligned free block of that size; on a fragmented >>>> aperture no such aligned block may exist even though enough contiguous >>>> space is free, so the allocation fails with -ENOSPC. >>>> >>>> The non-range contiguous path already recovers from this via >>>> __alloc_contig_try_harder(), which stitches an exact-size span from >>>> smaller adjacent blocks, but that fallback was unreachable once >>>> GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder() >>>> a [range_start, range_end) window and route the range + contiguous >>>> case through it: each candidate placement is confined to the window >>>> and aligned to min_block_size. Non-range callers pass [0, mm->size), >>>> where the guards are no-ops, so the existing behaviour is unchanged. >>>> A KUnit regression test covers the fragmentation pattern. >>>> >>>> Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b >>>> regression. >>>> >>>> v2: >>>> - Drop the split-undo patch; the range-bias search always >>>> descends to >>>> an exact-order block or fails the split (already handled), so the >>>> extra undo was redundant. (Matthew) >>>> - Verified this fallback alone fixes the kms_plane regression. >>>> >>>> Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with >>>> decoupled dirty tracker") >>> >>> Patch looks more like totally new functionally/feature, so the fixes >>> here is maybe unexpected? Did something in that fixes commit change >>> something such that the try_harder is now needed, but that needs >>> some expansion with bias + contig? Do we know exactly what changed >>> here? >> That commit removed __force_merge(), which previously helped recover >> larger contiguous allocations on demand. As a result, some range- > > __force_merge() got nuked, but IIRC I think that was essentially > because we now "force merge" on free, so shouldn't we get the ~same > result? Or is the fact that we force_merge() on every free giving a > different layout of pages, and with some very some specific allocation > pattern that difference subtly results in -ENOSPC, somehow? You are right - merge-on-free reclaims contiguity just like the old on-demand __force_merge() , so that is not the cause and the layout is effectively the same. The real change is where clear/dirty segregation lives: it used to be an inline guard inside __alloc_range_bias() , but the rework pulled it out into dirty_steer_window() , which only runs for non-range allocations. So range (aperture) requests now go through __alloc_range_bias() with no clear/dirty guard at all - clear and dirty allocations interleave and fragment the aperture until no large same-class run survives for a 4K scanout pin (-> -ENOSPC). Restoring that two-pass guard is the real fix. I have accordingly moved the Fixes: tag onto that guard-restore patch, and kept the range support in __alloc_contig_try_harder() as a new feature - it still helps, but it is a safety-net fallback that should land on top, not as the root-cause fix. Regards, Arun. > >> restricted contiguous allocation requests >> that used to succeed now fail with -ENOSPC (the kms_plane >> regression). This patch restores that capability through the >> try_harder() stitching logic, so the Fixes: tag reflects a >> regression introduced by that commit rather than new functionality. >>> >>>> Assisted-by: Claude:claude-opus-4-8 >>> >>> Assisted-by: LLM >>> >>>> Cc: Matthew Auld <matthew.auld@intel.com> >>>> Cc: Christian König <christian.koenig@amd.com> >>>> Signed-off-by: Arunpravin Paneer Selvam >>>> <Arunpravin.PaneerSelvam@amd.com> >>>> --- >>>> drivers/gpu/buddy.c | 88 >>>> ++++++++++++++++-------------- >>>> drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++-- >>>> 2 files changed, 126 insertions(+), 45 deletions(-) >>>> >>>> diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c >>>> index 2f2aaadafe35..5265e1f6a318 100644 >>>> --- a/drivers/gpu/buddy.c >>>> +++ b/drivers/gpu/buddy.c >>>> @@ -1685,26 +1685,14 @@ static int __gpu_buddy_alloc_range(struct >>>> gpu_buddy *mm, >>>> blocks, total_allocated_on_err); >>>> } >>>> -static int __alloc_contig_aligned_retry(struct gpu_buddy *mm, >>>> - u64 unaligned_offset, >>>> - u64 size, >>>> - u64 min_block_size, >>>> - unsigned long flags, >>>> - struct list_head *blocks) >>>> -{ >>>> - u64 aligned_offset = round_down(unaligned_offset, >>>> min_block_size); >>>> - >>>> - return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags, >>>> - NULL, blocks); >>>> -} >>>> - >>>> static int __alloc_contig_try_harder(struct gpu_buddy *mm, >>>> + u64 range_start, u64 range_end, >>>> u64 size, >>>> u64 min_block_size, >>>> unsigned long flags, >>>> struct list_head *blocks) >>>> { >>>> - u64 rhs_offset, lhs_offset, filled; >>>> + u64 rhs_offset, lhs_offset, filled, aligned; >>>> struct gpu_buddy_block *block; >>>> struct rb_root *root; >>>> struct rb_node *iter; >>>> @@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct >>>> gpu_buddy *mm, >>>> flags, &filled, blocks); >>>> if (err && err != -ENOSPC) >>>> return err; >>>> - if (!err && IS_ALIGNED(rhs_offset, min_block_size)) >>>> + if (!err && IS_ALIGNED(rhs_offset, min_block_size) && >>>> + rhs_offset >= range_start && rhs_offset + size <= >>>> range_end) >>>> return 0; >>>> if (!err) { >>>> /* Allocate the unaligned RHS offset using round_down */ >>>> gpu_buddy_free_list_internal(mm, blocks); >>>> - err = __alloc_contig_aligned_retry(mm, rhs_offset, >>>> - size, >>>> - min_block_size, >>>> - flags, blocks); >>>> - if (!err) >>>> - return 0; >>>> - if (err != -ENOSPC) { >>>> - gpu_buddy_free_list_internal(mm, blocks); >>>> - return err; >>>> + >>>> + aligned = round_down(rhs_offset, min_block_size); >>>> + if (aligned >= range_start && >>>> + aligned + size <= range_end) { >>>> + err = __gpu_buddy_alloc_range(mm, aligned, size, >>>> + flags, NULL, blocks); >>> >>> Did you consider doing a bias-range for [start, end] using >>> min_block_size and using what it returns as the starting point, >>> extending left/right? If that fails advance start and try again? >>> Maybe what you have here is much better for the case you have in mind? >> I took a closer look at the alloc_range_bias approach. My >> understanding is that this would repeatedly bias within [start, end], >> use the returned min_block_size block as a starting point, and then >> extend left/right to build the requested run. >> While that should work, every failed attempt may require splitting >> higher-order blocks to obtain a min_block_size block, followed by a >> free/merge back when the extension fails. >> The free-tree descent evaluates the available starting points through >> a tree walk without any splitting or allocation just for enumeration, >> so I kept that approach. >> Please let me know if there is a case where alloc_range_bias would >> provide an advantage over the free-tree descent. >> >> While evaluating it, I found two issues in the current try_harder() >> implementation: >> >> 1. It only searches free_tree[size_order]. This is insufficient when >> min_block_size < size. For example, for a 128 KiB allocation >> (order-5) with min_block_size = 4 KiB, free_tree[5] may be empty >> while two adjacent order-4 (64 KiB) blocks covering [64 KiB, 128 KiB] >> and [128 KiB, 192 KiB] are free. These blocks still form a valid >> order-5-sized contiguous run, but they can never appear in >> free_tree[5] because they are not mergeable buddies. As a result, a >> search that only considers free_tree[5] will never find this >> placement. v3 addresses this by walking the free tree from the >> requested size order down to min_order, while reusing the existing >> extend/stitch logic and min_block_size alignment checks unchanged. >> >> 2. The search is not restricted to the requested range. v3 confines >> the walk to [range_start, range_end], skipping candidates at or above >> range_end and stopping once the walk falls below range_start. >> This ensures that range-constrained allocations only consider free >> blocks within the requested window and avoids performing >> allocation/free attempts on blocks outside the specified range. >> For non-range callers, the effective window remains [0, mm->size], so >> the behavior is unchanged. >> >> Regards, >> Arun. >>> >>>> + if (!err) >>>> + return 0; >>>> + if (err != -ENOSPC) { >>>> + gpu_buddy_free_list_internal(mm, blocks); >>>> + return err; >>>> + } >>>> } >>>> goto next; >>>> } >>>> @@ -1759,15 +1751,17 @@ static int __alloc_contig_try_harder(struct >>>> gpu_buddy *mm, >>>> /* Allocate the unaligned LHS offset using round_down */ >>>> gpu_buddy_free_list_internal(mm, blocks); >>>> - err = __alloc_contig_aligned_retry(mm, lhs_offset, >>>> - size, >>>> - min_block_size, >>>> - flags, blocks); >>>> - if (!err) >>>> - return 0; >>>> - if (err != -ENOSPC) { >>>> - gpu_buddy_free_list_internal(mm, blocks); >>>> - return err; >>>> + >>>> + aligned = round_down(lhs_offset, min_block_size); >>>> + if (aligned >= range_start && aligned + size <= range_end) { >>>> + err = __gpu_buddy_alloc_range(mm, aligned, size, >>>> + flags, NULL, blocks); >>>> + if (!err) >>>> + return 0; >>>> + if (err != -ENOSPC) { >>>> + gpu_buddy_free_list_internal(mm, blocks); >>>> + return err; >>>> + } >>>> } >>>> next: >>>> gpu_buddy_free_list_internal(mm, blocks); >>>> @@ -2021,11 +2015,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy >>>> *mm, >>>> min_order = ilog2(min_block_size) - ilog2(mm->chunk_size); >>>> if (order > mm->max_order || size > mm->size) { >>>> - if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) && >>>> - !(flags & GPU_BUDDY_RANGE_ALLOCATION)) >>>> - return __alloc_contig_try_harder(mm, original_size, >>>> + if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) { >>>> + u64 range_start, range_end; >>>> + >>>> + range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? >>>> start : 0; >>>> + range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end >>>> : mm->size; >>>> + >>>> + return __alloc_contig_try_harder(mm, range_start, >>>> + range_end, >>>> + original_size, >>>> original_min_size, >>>> flags, blocks); >>>> + } >>>> return -EINVAL; >>>> } >>>> @@ -2058,9 +2059,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy >>>> *mm, >>>> * Try contiguous block allocation through >>>> * try harder method. >>>> */ >>>> - if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION && >>>> - !(flags & GPU_BUDDY_RANGE_ALLOCATION)) { >>>> - err = __alloc_contig_try_harder(mm, >>>> + if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) { >>>> + u64 range_start, range_end; >>>> + >>>> + range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) >>>> ? start : 0; >>>> + range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? >>>> end : mm->size; >>>> + >>>> + err = __alloc_contig_try_harder(mm, range_start, >>>> + range_end, >>>> original_size, >>>> original_min_size, >>>> flags, >>>> @@ -2068,9 +2074,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm, >>>> if (!err) >>>> return 0; >>>> if (err != -ENOSPC) >>>> - return err; >>>> - goto err_free; >>>> + goto err_free; >>>> } >>>> + >>>> err = -ENOSPC; >>>> goto err_free; >>>> } while (1); >>>> diff --git a/drivers/gpu/tests/gpu_buddy_test.c >>>> b/drivers/gpu/tests/ gpu_buddy_test.c >>>> index b75d32ca6ca0..2c445870b808 100644 >>>> --- a/drivers/gpu/tests/gpu_buddy_test.c >>>> +++ b/drivers/gpu/tests/gpu_buddy_test.c >>>> @@ -1251,6 +1251,77 @@ static void >>>> gpu_test_buddy_alloc_contiguous(struct kunit *test) >>>> gpu_buddy_fini(&mm); >>>> } >>>> +static void gpu_test_buddy_alloc_range_contiguous(struct kunit >>>> *test) >>>> +{ >>>> + const unsigned long ps = SZ_4K, mm_size = 16 * ps; >>>> + const unsigned long range_end = 8 * ps; >>>> + struct gpu_buddy_block *block, *prev; >>>> + LIST_HEAD(allocated); >>>> + struct gpu_buddy mm; >>>> + LIST_HEAD(pin_lo); >>>> + LIST_HEAD(pin_hi); >>>> + u64 total; >>>> + >>>> + KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps), >>>> + "buddy_init failed\n"); >>>> + >>>> + /* >>>> + * Idea is to confine the test to the sub-range [0, 32K), >>>> which a 12K >>>> + * contiguous request (rounded up to 16K) splits into two >>>> naturally >>>> + * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the >>>> first 4K page >>>> + * of each slot ([0, 4K) and [16K, 20K)) so that neither >>>> aligned slot >>>> + * can satisfy the rounded-up 16K allocation, yet the freed >>>> remainder >>>> + * still leaves a contiguous 12K hole at offset 4K, which is >>>> page-aligned >>>> + * but not 16K-aligned. A 12K contiguous+range allocation must >>>> therefore >>>> + * fall back to stitching that span instead of returning -ENOSPC. >>>> + */ >>>> + KUNIT_ASSERT_FALSE_MSG(test, >>>> + gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps, >>>> + &pin_lo, 0), >>>> + "failed to pin low page\n"); >>>> + KUNIT_ASSERT_FALSE_MSG(test, >>>> + gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps, >>>> + ps, &pin_hi, 0), >>>> + "failed to pin high page\n"); >>>> + >>>> + /* No aligned 16K block is free; the range-aware fallback must >>>> stitch >>>> + * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC. >>>> + */ >>>> + KUNIT_ASSERT_FALSE_MSG(test, >>>> + gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps, >>>> + ps, &allocated, >>>> + GPU_BUDDY_CONTIGUOUS_ALLOCATION | >>>> + GPU_BUDDY_RANGE_ALLOCATION), >>>> + "range-restricted contiguous alloc failed\n"); >>>> + >>>> + /* The result must be exactly 3*ps, contiguous, and inside the >>>> range. */ >>>> + total = 0; >>>> + prev = NULL; >>>> + list_for_each_entry(block, &allocated, link) { >>>> + u64 offset = gpu_buddy_block_offset(block); >>>> + u64 bsize = gpu_buddy_block_size(&mm, block); >>>> + >>>> + KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end, >>>> + "block [%llx, %llx) outside range\n", >>>> + offset, offset + bsize); >>>> + if (prev) >>>> + KUNIT_EXPECT_EQ_MSG(test, >>>> + gpu_buddy_block_offset(prev) + >>>> + gpu_buddy_block_size(&mm, prev), >>>> + offset, >>>> + "block at %llx not contiguous\n", >>>> + offset); >>>> + prev = block; >>>> + total += bsize; >>>> + } >>>> + KUNIT_EXPECT_EQ(test, total, 3 * ps); >>>> + >>>> + gpu_buddy_free_list(&mm, &allocated, 0); >>>> + gpu_buddy_free_list(&mm, &pin_lo, 0); >>>> + gpu_buddy_free_list(&mm, &pin_hi, 0); >>>> + gpu_buddy_fini(&mm); >>>> +} >>>> + >>>> static void gpu_test_buddy_alloc_pathological(struct kunit *test) >>>> { >>>> u64 mm_size, size, start = 0; >>>> @@ -1534,10 +1605,13 @@ static void >>>> gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test) >>>> GPU_BUDDY_RANGE_ALLOCATION); >>>> KUNIT_EXPECT_EQ(test, err, -EINVAL); >>>> - /* CONTIGUOUS + RANGE should return -EINVAL (no try_harder >>>> for RANGE) */ >>>> - err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, >>>> &blocks, >>>> - GPU_BUDDY_CONTIGUOUS_ALLOCATION | >>>> GPU_BUDDY_RANGE_ALLOCATION); >>>> - KUNIT_EXPECT_EQ(test, err, -EINVAL); >>>> + /* CONTIGUOUS + RANGE should succeed via the range-aware >>>> try_harder */ >>>> + KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, >>>> mm_size, size, >>>> + SZ_4K, &blocks, >>>> + GPU_BUDDY_CONTIGUOUS_ALLOCATION | >>>> + GPU_BUDDY_RANGE_ALLOCATION), >>>> + "range contiguous alloc hit an error >>>> size=%llu\n", size); >>>> + gpu_buddy_free_list(&mm, &blocks, 0); >>>> gpu_buddy_fini(&mm); >>>> } >>>> @@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = { >>>> KUNIT_CASE(gpu_test_buddy_alloc_pessimistic), >>>> KUNIT_CASE(gpu_test_buddy_alloc_pathological), >>>> KUNIT_CASE(gpu_test_buddy_alloc_contiguous), >>>> + KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous), >>>> KUNIT_CASE(gpu_test_buddy_alloc_clear), >>>> KUNIT_CASE(gpu_test_buddy_alloc_range), >>>> KUNIT_CASE(gpu_test_buddy_alloc_range_bias), >>>> >>>> base-commit: 744f262401cc2e1f3827496c72a47e072d31a852 >>> >> > ^ permalink raw reply [flat|nested] 8+ messages in thread
* Re: [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback 2026-10-09 9:14 ` Arunpravin Paneer Selvam @ 2026-10-09 10:17 ` Matthew Auld 0 siblings, 0 replies; 8+ messages in thread From: Matthew Auld @ 2026-10-09 10:17 UTC (permalink / raw) To: Arunpravin Paneer Selvam, dri-devel, intel-gfx, intel-xe, amd-gfx Cc: christian.koenig, alexander.deucher, Anand.Raghavendra On 09/10/2026 10:14, Arunpravin Paneer Selvam wrote: > > > On 10/6/2026 3:27 PM, Matthew Auld wrote: >> On 05/10/2026 15:14, Arunpravin Paneer Selvam wrote: >>> >>> >>> On 10/1/2026 11:58 PM, Matthew Auld wrote: >>>> On 30/09/2026 07:21, Arunpravin Paneer Selvam wrote: >>>>> From: Arunpravin Paneer Selvam <Arunpravin.PaneerSelvam@amd.com> >>>>> >>>>> A range + contiguous allocation (e.g. a scanout FB confined to the >>>>> CPU-visible VRAM aperture) rounds its size up to a power of two and >>>>> requires a naturally aligned free block of that size; on a fragmented >>>>> aperture no such aligned block may exist even though enough contiguous >>>>> space is free, so the allocation fails with -ENOSPC. >>>>> >>>>> The non-range contiguous path already recovers from this via >>>>> __alloc_contig_try_harder(), which stitches an exact-size span from >>>>> smaller adjacent blocks, but that fallback was unreachable once >>>>> GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder() >>>>> a [range_start, range_end) window and route the range + contiguous >>>>> case through it: each candidate placement is confined to the window >>>>> and aligned to min_block_size. Non-range callers pass [0, mm->size), >>>>> where the guards are no-ops, so the existing behaviour is unchanged. >>>>> A KUnit regression test covers the fragmentation pattern. >>>>> >>>>> Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b >>>>> regression. >>>>> >>>>> v2: >>>>> - Drop the split-undo patch; the range-bias search always >>>>> descends to >>>>> an exact-order block or fails the split (already handled), so the >>>>> extra undo was redundant. (Matthew) >>>>> - Verified this fallback alone fixes the kms_plane regression. >>>>> >>>>> Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with >>>>> decoupled dirty tracker") >>>> >>>> Patch looks more like totally new functionally/feature, so the fixes >>>> here is maybe unexpected? Did something in that fixes commit change >>>> something such that the try_harder is now needed, but that needs >>>> some expansion with bias + contig? Do we know exactly what changed >>>> here? >>> That commit removed __force_merge(), which previously helped recover >>> larger contiguous allocations on demand. As a result, some range- >> >> __force_merge() got nuked, but IIRC I think that was essentially >> because we now "force merge" on free, so shouldn't we get the ~same >> result? Or is the fact that we force_merge() on every free giving a >> different layout of pages, and with some very some specific allocation >> pattern that difference subtly results in -ENOSPC, somehow? > You are right - merge-on-free reclaims contiguity just like the old on- > demand __force_merge() , so that is not the cause and the layout is > effectively the same. The real change is where clear/dirty segregation > lives: it used to be an inline guard inside __alloc_range_bias() , but > the rework pulled it out into dirty_steer_window() , which only runs for > non-range allocations. So range (aperture) requests now go through > __alloc_range_bias() with no clear/dirty guard at all - clear and dirty > allocations interleave and fragment the aperture until no large same- > class run survives for a 4K scanout pin (-> -ENOSPC). Restoring that > two-pass guard is the real fix. Ahh, yeah that makes more sense now :) > > I have accordingly moved the Fixes: tag onto that guard-restore patch, > and kept the range support in __alloc_contig_try_harder() as a new > feature - it still helps, but it is a safety-net fallback that should > land on top, not as the root-cause fix. > > Regards, > Arun. >> >>> restricted contiguous allocation requests >>> that used to succeed now fail with -ENOSPC (the kms_plane >>> regression). This patch restores that capability through the >>> try_harder() stitching logic, so the Fixes: tag reflects a >>> regression introduced by that commit rather than new functionality. >>>> >>>>> Assisted-by: Claude:claude-opus-4-8 >>>> >>>> Assisted-by: LLM >>>> >>>>> Cc: Matthew Auld <matthew.auld@intel.com> >>>>> Cc: Christian König <christian.koenig@amd.com> >>>>> Signed-off-by: Arunpravin Paneer Selvam >>>>> <Arunpravin.PaneerSelvam@amd.com> >>>>> --- >>>>> drivers/gpu/buddy.c | 88 +++++++++++++++ >>>>> +-------------- >>>>> drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++-- >>>>> 2 files changed, 126 insertions(+), 45 deletions(-) >>>>> >>>>> diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c >>>>> index 2f2aaadafe35..5265e1f6a318 100644 >>>>> --- a/drivers/gpu/buddy.c >>>>> +++ b/drivers/gpu/buddy.c >>>>> @@ -1685,26 +1685,14 @@ static int __gpu_buddy_alloc_range(struct >>>>> gpu_buddy *mm, >>>>> blocks, total_allocated_on_err); >>>>> } >>>>> -static int __alloc_contig_aligned_retry(struct gpu_buddy *mm, >>>>> - u64 unaligned_offset, >>>>> - u64 size, >>>>> - u64 min_block_size, >>>>> - unsigned long flags, >>>>> - struct list_head *blocks) >>>>> -{ >>>>> - u64 aligned_offset = round_down(unaligned_offset, >>>>> min_block_size); >>>>> - >>>>> - return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags, >>>>> - NULL, blocks); >>>>> -} >>>>> - >>>>> static int __alloc_contig_try_harder(struct gpu_buddy *mm, >>>>> + u64 range_start, u64 range_end, >>>>> u64 size, >>>>> u64 min_block_size, >>>>> unsigned long flags, >>>>> struct list_head *blocks) >>>>> { >>>>> - u64 rhs_offset, lhs_offset, filled; >>>>> + u64 rhs_offset, lhs_offset, filled, aligned; >>>>> struct gpu_buddy_block *block; >>>>> struct rb_root *root; >>>>> struct rb_node *iter; >>>>> @@ -1734,20 +1722,24 @@ static int __alloc_contig_try_harder(struct >>>>> gpu_buddy *mm, >>>>> flags, &filled, blocks); >>>>> if (err && err != -ENOSPC) >>>>> return err; >>>>> - if (!err && IS_ALIGNED(rhs_offset, min_block_size)) >>>>> + if (!err && IS_ALIGNED(rhs_offset, min_block_size) && >>>>> + rhs_offset >= range_start && rhs_offset + size <= >>>>> range_end) >>>>> return 0; >>>>> if (!err) { >>>>> /* Allocate the unaligned RHS offset using round_down */ >>>>> gpu_buddy_free_list_internal(mm, blocks); >>>>> - err = __alloc_contig_aligned_retry(mm, rhs_offset, >>>>> - size, >>>>> - min_block_size, >>>>> - flags, blocks); >>>>> - if (!err) >>>>> - return 0; >>>>> - if (err != -ENOSPC) { >>>>> - gpu_buddy_free_list_internal(mm, blocks); >>>>> - return err; >>>>> + >>>>> + aligned = round_down(rhs_offset, min_block_size); >>>>> + if (aligned >= range_start && >>>>> + aligned + size <= range_end) { >>>>> + err = __gpu_buddy_alloc_range(mm, aligned, size, >>>>> + flags, NULL, blocks); >>>> >>>> Did you consider doing a bias-range for [start, end] using >>>> min_block_size and using what it returns as the starting point, >>>> extending left/right? If that fails advance start and try again? >>>> Maybe what you have here is much better for the case you have in mind? >>> I took a closer look at the alloc_range_bias approach. My >>> understanding is that this would repeatedly bias within [start, end], >>> use the returned min_block_size block as a starting point, and then >>> extend left/right to build the requested run. >>> While that should work, every failed attempt may require splitting >>> higher-order blocks to obtain a min_block_size block, followed by a >>> free/merge back when the extension fails. >>> The free-tree descent evaluates the available starting points through >>> a tree walk without any splitting or allocation just for enumeration, >>> so I kept that approach. >>> Please let me know if there is a case where alloc_range_bias would >>> provide an advantage over the free-tree descent. >>> >>> While evaluating it, I found two issues in the current try_harder() >>> implementation: >>> >>> 1. It only searches free_tree[size_order]. This is insufficient when >>> min_block_size < size. For example, for a 128 KiB allocation >>> (order-5) with min_block_size = 4 KiB, free_tree[5] may be empty >>> while two adjacent order-4 (64 KiB) blocks covering [64 KiB, 128 KiB] >>> and [128 KiB, 192 KiB] are free. These blocks still form a valid >>> order-5-sized contiguous run, but they can never appear in >>> free_tree[5] because they are not mergeable buddies. As a result, a >>> search that only considers free_tree[5] will never find this >>> placement. v3 addresses this by walking the free tree from the >>> requested size order down to min_order, while reusing the existing >>> extend/stitch logic and min_block_size alignment checks unchanged. >>> >>> 2. The search is not restricted to the requested range. v3 confines >>> the walk to [range_start, range_end], skipping candidates at or above >>> range_end and stopping once the walk falls below range_start. >>> This ensures that range-constrained allocations only consider free >>> blocks within the requested window and avoids performing allocation/ >>> free attempts on blocks outside the specified range. >>> For non-range callers, the effective window remains [0, mm->size], so >>> the behavior is unchanged. >>> >>> Regards, >>> Arun. >>>> >>>>> + if (!err) >>>>> + return 0; >>>>> + if (err != -ENOSPC) { >>>>> + gpu_buddy_free_list_internal(mm, blocks); >>>>> + return err; >>>>> + } >>>>> } >>>>> goto next; >>>>> } >>>>> @@ -1759,15 +1751,17 @@ static int __alloc_contig_try_harder(struct >>>>> gpu_buddy *mm, >>>>> /* Allocate the unaligned LHS offset using round_down */ >>>>> gpu_buddy_free_list_internal(mm, blocks); >>>>> - err = __alloc_contig_aligned_retry(mm, lhs_offset, >>>>> - size, >>>>> - min_block_size, >>>>> - flags, blocks); >>>>> - if (!err) >>>>> - return 0; >>>>> - if (err != -ENOSPC) { >>>>> - gpu_buddy_free_list_internal(mm, blocks); >>>>> - return err; >>>>> + >>>>> + aligned = round_down(lhs_offset, min_block_size); >>>>> + if (aligned >= range_start && aligned + size <= range_end) { >>>>> + err = __gpu_buddy_alloc_range(mm, aligned, size, >>>>> + flags, NULL, blocks); >>>>> + if (!err) >>>>> + return 0; >>>>> + if (err != -ENOSPC) { >>>>> + gpu_buddy_free_list_internal(mm, blocks); >>>>> + return err; >>>>> + } >>>>> } >>>>> next: >>>>> gpu_buddy_free_list_internal(mm, blocks); >>>>> @@ -2021,11 +2015,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy >>>>> *mm, >>>>> min_order = ilog2(min_block_size) - ilog2(mm->chunk_size); >>>>> if (order > mm->max_order || size > mm->size) { >>>>> - if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) && >>>>> - !(flags & GPU_BUDDY_RANGE_ALLOCATION)) >>>>> - return __alloc_contig_try_harder(mm, original_size, >>>>> + if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) { >>>>> + u64 range_start, range_end; >>>>> + >>>>> + range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? >>>>> start : 0; >>>>> + range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? >>>>> end : mm->size; >>>>> + >>>>> + return __alloc_contig_try_harder(mm, range_start, >>>>> + range_end, >>>>> + original_size, >>>>> original_min_size, >>>>> flags, blocks); >>>>> + } >>>>> return -EINVAL; >>>>> } >>>>> @@ -2058,9 +2059,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy >>>>> *mm, >>>>> * Try contiguous block allocation through >>>>> * try harder method. >>>>> */ >>>>> - if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION && >>>>> - !(flags & GPU_BUDDY_RANGE_ALLOCATION)) { >>>>> - err = __alloc_contig_try_harder(mm, >>>>> + if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) { >>>>> + u64 range_start, range_end; >>>>> + >>>>> + range_start = (flags & >>>>> GPU_BUDDY_RANGE_ALLOCATION) ? start : 0; >>>>> + range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? >>>>> end : mm->size; >>>>> + >>>>> + err = __alloc_contig_try_harder(mm, range_start, >>>>> + range_end, >>>>> original_size, >>>>> original_min_size, >>>>> flags, >>>>> @@ -2068,9 +2074,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm, >>>>> if (!err) >>>>> return 0; >>>>> if (err != -ENOSPC) >>>>> - return err; >>>>> - goto err_free; >>>>> + goto err_free; >>>>> } >>>>> + >>>>> err = -ENOSPC; >>>>> goto err_free; >>>>> } while (1); >>>>> diff --git a/drivers/gpu/tests/gpu_buddy_test.c b/drivers/gpu/ >>>>> tests/ gpu_buddy_test.c >>>>> index b75d32ca6ca0..2c445870b808 100644 >>>>> --- a/drivers/gpu/tests/gpu_buddy_test.c >>>>> +++ b/drivers/gpu/tests/gpu_buddy_test.c >>>>> @@ -1251,6 +1251,77 @@ static void >>>>> gpu_test_buddy_alloc_contiguous(struct kunit *test) >>>>> gpu_buddy_fini(&mm); >>>>> } >>>>> +static void gpu_test_buddy_alloc_range_contiguous(struct kunit >>>>> *test) >>>>> +{ >>>>> + const unsigned long ps = SZ_4K, mm_size = 16 * ps; >>>>> + const unsigned long range_end = 8 * ps; >>>>> + struct gpu_buddy_block *block, *prev; >>>>> + LIST_HEAD(allocated); >>>>> + struct gpu_buddy mm; >>>>> + LIST_HEAD(pin_lo); >>>>> + LIST_HEAD(pin_hi); >>>>> + u64 total; >>>>> + >>>>> + KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps), >>>>> + "buddy_init failed\n"); >>>>> + >>>>> + /* >>>>> + * Idea is to confine the test to the sub-range [0, 32K), >>>>> which a 12K >>>>> + * contiguous request (rounded up to 16K) splits into two >>>>> naturally >>>>> + * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the >>>>> first 4K page >>>>> + * of each slot ([0, 4K) and [16K, 20K)) so that neither >>>>> aligned slot >>>>> + * can satisfy the rounded-up 16K allocation, yet the freed >>>>> remainder >>>>> + * still leaves a contiguous 12K hole at offset 4K, which is >>>>> page-aligned >>>>> + * but not 16K-aligned. A 12K contiguous+range allocation must >>>>> therefore >>>>> + * fall back to stitching that span instead of returning -ENOSPC. >>>>> + */ >>>>> + KUNIT_ASSERT_FALSE_MSG(test, >>>>> + gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps, >>>>> + &pin_lo, 0), >>>>> + "failed to pin low page\n"); >>>>> + KUNIT_ASSERT_FALSE_MSG(test, >>>>> + gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps, >>>>> + ps, &pin_hi, 0), >>>>> + "failed to pin high page\n"); >>>>> + >>>>> + /* No aligned 16K block is free; the range-aware fallback must >>>>> stitch >>>>> + * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC. >>>>> + */ >>>>> + KUNIT_ASSERT_FALSE_MSG(test, >>>>> + gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps, >>>>> + ps, &allocated, >>>>> + GPU_BUDDY_CONTIGUOUS_ALLOCATION | >>>>> + GPU_BUDDY_RANGE_ALLOCATION), >>>>> + "range-restricted contiguous alloc failed\n"); >>>>> + >>>>> + /* The result must be exactly 3*ps, contiguous, and inside the >>>>> range. */ >>>>> + total = 0; >>>>> + prev = NULL; >>>>> + list_for_each_entry(block, &allocated, link) { >>>>> + u64 offset = gpu_buddy_block_offset(block); >>>>> + u64 bsize = gpu_buddy_block_size(&mm, block); >>>>> + >>>>> + KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end, >>>>> + "block [%llx, %llx) outside range\n", >>>>> + offset, offset + bsize); >>>>> + if (prev) >>>>> + KUNIT_EXPECT_EQ_MSG(test, >>>>> + gpu_buddy_block_offset(prev) + >>>>> + gpu_buddy_block_size(&mm, prev), >>>>> + offset, >>>>> + "block at %llx not contiguous\n", >>>>> + offset); >>>>> + prev = block; >>>>> + total += bsize; >>>>> + } >>>>> + KUNIT_EXPECT_EQ(test, total, 3 * ps); >>>>> + >>>>> + gpu_buddy_free_list(&mm, &allocated, 0); >>>>> + gpu_buddy_free_list(&mm, &pin_lo, 0); >>>>> + gpu_buddy_free_list(&mm, &pin_hi, 0); >>>>> + gpu_buddy_fini(&mm); >>>>> +} >>>>> + >>>>> static void gpu_test_buddy_alloc_pathological(struct kunit *test) >>>>> { >>>>> u64 mm_size, size, start = 0; >>>>> @@ -1534,10 +1605,13 @@ static void >>>>> gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test) >>>>> GPU_BUDDY_RANGE_ALLOCATION); >>>>> KUNIT_EXPECT_EQ(test, err, -EINVAL); >>>>> - /* CONTIGUOUS + RANGE should return -EINVAL (no try_harder >>>>> for RANGE) */ >>>>> - err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, >>>>> &blocks, >>>>> - GPU_BUDDY_CONTIGUOUS_ALLOCATION | >>>>> GPU_BUDDY_RANGE_ALLOCATION); >>>>> - KUNIT_EXPECT_EQ(test, err, -EINVAL); >>>>> + /* CONTIGUOUS + RANGE should succeed via the range-aware >>>>> try_harder */ >>>>> + KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, >>>>> mm_size, size, >>>>> + SZ_4K, &blocks, >>>>> + GPU_BUDDY_CONTIGUOUS_ALLOCATION | >>>>> + GPU_BUDDY_RANGE_ALLOCATION), >>>>> + "range contiguous alloc hit an error >>>>> size=%llu\n", size); >>>>> + gpu_buddy_free_list(&mm, &blocks, 0); >>>>> gpu_buddy_fini(&mm); >>>>> } >>>>> @@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = { >>>>> KUNIT_CASE(gpu_test_buddy_alloc_pessimistic), >>>>> KUNIT_CASE(gpu_test_buddy_alloc_pathological), >>>>> KUNIT_CASE(gpu_test_buddy_alloc_contiguous), >>>>> + KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous), >>>>> KUNIT_CASE(gpu_test_buddy_alloc_clear), >>>>> KUNIT_CASE(gpu_test_buddy_alloc_range), >>>>> KUNIT_CASE(gpu_test_buddy_alloc_range_bias), >>>>> >>>>> base-commit: 744f262401cc2e1f3827496c72a47e072d31a852 >>>> >>> >> > ^ permalink raw reply [flat|nested] 8+ messages in thread
end of thread, other threads:[~2026-10-09 10:17 UTC | newest] Thread overview: 8+ messages (download: mbox.gz follow: Atom feed -- links below jump to the message on this page -- 2026-09-30 6:21 [PATCH v2] gpu/buddy: add range-restricted contiguous allocation fallback Arunpravin Paneer Selvam 2026-09-30 6:31 ` sashiko-bot 2026-10-01 7:06 ` Arunpravin Paneer Selvam 2026-10-01 18:28 ` Matthew Auld 2026-10-05 14:14 ` Arunpravin Paneer Selvam 2026-10-06 9:57 ` Matthew Auld 2026-10-09 9:14 ` Arunpravin Paneer Selvam 2026-10-09 10:17 ` Matthew Auld
This is a public inbox, see mirroring instructions for how to clone and mirror all data and code used for this inbox