From: Arunpravin Paneer Selvam <[email protected]>
A range + contiguous allocation (e.g. a scanout FB confined to the
CPU-visible VRAM aperture) rounds its size up to a power of two and
requires a naturally aligned free block of that size; on a fragmented
aperture no such aligned block may exist even though enough contiguous
space is free, so the allocation fails with -ENOSPC.
The non-range contiguous path already recovers from this via
__alloc_contig_try_harder(), which stitches an exact-size span from
smaller adjacent blocks, but that fallback was unreachable once
GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder()
a [range_start, range_end) window and route the range + contiguous
case through it: each candidate placement is confined to the window
and aligned to min_block_size. Non-range callers pass [0, mm->size),
where the guards are no-ops, so the existing behaviour is unchanged.
A KUnit regression test covers the fragmentation pattern.
Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b
regression.
Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with decoupled
dirty tracker")
Assisted-by: Claude:claude-opus-4-8
Cc: Matthew Auld <[email protected]>
Cc: Christian König <[email protected]>
Signed-off-by: Arunpravin Paneer Selvam <[email protected]>
---
drivers/gpu/buddy.c | 88 ++++++++++++++++--------------
drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++--
2 files changed, 126 insertions(+), 45 deletions(-)
diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c
index e5c9e21cd077..08e82378a9f0 100644
--- a/drivers/gpu/buddy.c
+++ b/drivers/gpu/buddy.c
@@ -1735,26 +1735,14 @@ static int __gpu_buddy_alloc_range(struct gpu_buddy *mm,
blocks, total_allocated_on_err);
}
-static int __alloc_contig_aligned_retry(struct gpu_buddy *mm,
- u64 unaligned_offset,
- u64 size,
- u64 min_block_size,
- unsigned long flags,
- struct list_head *blocks)
-{
- u64 aligned_offset = round_down(unaligned_offset, min_block_size);
-
- return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags,
- NULL, blocks);
-}
-
static int __alloc_contig_try_harder(struct gpu_buddy *mm,
+ u64 range_start, u64 range_end,
u64 size,
u64 min_block_size,
unsigned long flags,
struct list_head *blocks)
{
- u64 rhs_offset, lhs_offset, filled;
+ u64 rhs_offset, lhs_offset, filled, aligned;
struct gpu_buddy_block *block;
struct rb_root *root;
struct rb_node *iter;
@@ -1784,20 +1772,24 @@ static int __alloc_contig_try_harder(struct gpu_buddy
*mm,
flags, &filled, blocks);
if (err && err != -ENOSPC)
return err;
- if (!err && IS_ALIGNED(rhs_offset, min_block_size))
+ if (!err && IS_ALIGNED(rhs_offset, min_block_size) &&
+ rhs_offset >= range_start && rhs_offset + size <= range_end)
return 0;
if (!err) {
/* Allocate the unaligned RHS offset using round_down */
gpu_buddy_free_list_internal(mm, blocks);
- err = __alloc_contig_aligned_retry(mm, rhs_offset,
- size,
- min_block_size,
- flags, blocks);
- if (!err)
- return 0;
- if (err != -ENOSPC) {
- gpu_buddy_free_list_internal(mm, blocks);
- return err;
+
+ aligned = round_down(rhs_offset, min_block_size);
+ if (aligned >= range_start &&
+ aligned + size <= range_end) {
+ err = __gpu_buddy_alloc_range(mm, aligned, size,
+ flags, NULL,
blocks);
+ if (!err)
+ return 0;
+ if (err != -ENOSPC) {
+ gpu_buddy_free_list_internal(mm,
blocks);
+ return err;
+ }
}
goto next;
}
@@ -1809,15 +1801,17 @@ static int __alloc_contig_try_harder(struct gpu_buddy
*mm,
/* Allocate the unaligned LHS offset using round_down */
gpu_buddy_free_list_internal(mm, blocks);
- err = __alloc_contig_aligned_retry(mm, lhs_offset,
- size,
- min_block_size,
- flags, blocks);
- if (!err)
- return 0;
- if (err != -ENOSPC) {
- gpu_buddy_free_list_internal(mm, blocks);
- return err;
+
+ aligned = round_down(lhs_offset, min_block_size);
+ if (aligned >= range_start && aligned + size <= range_end) {
+ err = __gpu_buddy_alloc_range(mm, aligned, size,
+ flags, NULL, blocks);
+ if (!err)
+ return 0;
+ if (err != -ENOSPC) {
+ gpu_buddy_free_list_internal(mm, blocks);
+ return err;
+ }
}
next:
gpu_buddy_free_list_internal(mm, blocks);
@@ -2071,11 +2065,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
min_order = ilog2(min_block_size) - ilog2(mm->chunk_size);
if (order > mm->max_order || size > mm->size) {
- if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) &&
- !(flags & GPU_BUDDY_RANGE_ALLOCATION))
- return __alloc_contig_try_harder(mm, original_size,
+ if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
+ u64 range_start, range_end;
+
+ range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ?
start : 0;
+ range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end
: mm->size;
+
+ return __alloc_contig_try_harder(mm, range_start,
+ range_end,
+ original_size,
original_min_size,
flags, blocks);
+ }
return -EINVAL;
}
@@ -2108,9 +2109,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
* Try contiguous block allocation through
* try harder method.
*/
- if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION &&
- !(flags & GPU_BUDDY_RANGE_ALLOCATION)) {
- err = __alloc_contig_try_harder(mm,
+ if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
+ u64 range_start, range_end;
+
+ range_start = (flags &
GPU_BUDDY_RANGE_ALLOCATION) ? start : 0;
+ range_end = (flags &
GPU_BUDDY_RANGE_ALLOCATION) ? end : mm->size;
+
+ err = __alloc_contig_try_harder(mm, range_start,
+ range_end,
original_size,
original_min_size,
flags,
@@ -2118,9 +2124,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
if (!err)
return 0;
if (err != -ENOSPC)
- return err;
- goto err_free;
+ goto err_free;
}
+
err = -ENOSPC;
goto err_free;
} while (1);
diff --git a/drivers/gpu/tests/gpu_buddy_test.c
b/drivers/gpu/tests/gpu_buddy_test.c
index b75d32ca6ca0..2c445870b808 100644
--- a/drivers/gpu/tests/gpu_buddy_test.c
+++ b/drivers/gpu/tests/gpu_buddy_test.c
@@ -1251,6 +1251,77 @@ static void gpu_test_buddy_alloc_contiguous(struct kunit
*test)
gpu_buddy_fini(&mm);
}
+static void gpu_test_buddy_alloc_range_contiguous(struct kunit *test)
+{
+ const unsigned long ps = SZ_4K, mm_size = 16 * ps;
+ const unsigned long range_end = 8 * ps;
+ struct gpu_buddy_block *block, *prev;
+ LIST_HEAD(allocated);
+ struct gpu_buddy mm;
+ LIST_HEAD(pin_lo);
+ LIST_HEAD(pin_hi);
+ u64 total;
+
+ KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps),
+ "buddy_init failed\n");
+
+ /*
+ * Idea is to confine the test to the sub-range [0, 32K), which a 12K
+ * contiguous request (rounded up to 16K) splits into two naturally
+ * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the first 4K page
+ * of each slot ([0, 4K) and [16K, 20K)) so that neither aligned slot
+ * can satisfy the rounded-up 16K allocation, yet the freed remainder
+ * still leaves a contiguous 12K hole at offset 4K, which is
page-aligned
+ * but not 16K-aligned. A 12K contiguous+range allocation must therefore
+ * fall back to stitching that span instead of returning -ENOSPC.
+ */
+ KUNIT_ASSERT_FALSE_MSG(test,
+ gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps,
+ &pin_lo, 0),
+ "failed to pin low page\n");
+ KUNIT_ASSERT_FALSE_MSG(test,
+ gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps,
+ ps, &pin_hi, 0),
+ "failed to pin high page\n");
+
+ /* No aligned 16K block is free; the range-aware fallback must stitch
+ * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC.
+ */
+ KUNIT_ASSERT_FALSE_MSG(test,
+ gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps,
+ ps, &allocated,
+
GPU_BUDDY_CONTIGUOUS_ALLOCATION |
+
GPU_BUDDY_RANGE_ALLOCATION),
+ "range-restricted contiguous alloc failed\n");
+
+ /* The result must be exactly 3*ps, contiguous, and inside the range. */
+ total = 0;
+ prev = NULL;
+ list_for_each_entry(block, &allocated, link) {
+ u64 offset = gpu_buddy_block_offset(block);
+ u64 bsize = gpu_buddy_block_size(&mm, block);
+
+ KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end,
+ "block [%llx, %llx) outside range\n",
+ offset, offset + bsize);
+ if (prev)
+ KUNIT_EXPECT_EQ_MSG(test,
+ gpu_buddy_block_offset(prev) +
+ gpu_buddy_block_size(&mm, prev),
+ offset,
+ "block at %llx not contiguous\n",
+ offset);
+ prev = block;
+ total += bsize;
+ }
+ KUNIT_EXPECT_EQ(test, total, 3 * ps);
+
+ gpu_buddy_free_list(&mm, &allocated, 0);
+ gpu_buddy_free_list(&mm, &pin_lo, 0);
+ gpu_buddy_free_list(&mm, &pin_hi, 0);
+ gpu_buddy_fini(&mm);
+}
+
static void gpu_test_buddy_alloc_pathological(struct kunit *test)
{
u64 mm_size, size, start = 0;
@@ -1534,10 +1605,13 @@ static void
gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test)
GPU_BUDDY_RANGE_ALLOCATION);
KUNIT_EXPECT_EQ(test, err, -EINVAL);
- /* CONTIGUOUS + RANGE should return -EINVAL (no try_harder for RANGE) */
- err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, &blocks,
- GPU_BUDDY_CONTIGUOUS_ALLOCATION |
GPU_BUDDY_RANGE_ALLOCATION);
- KUNIT_EXPECT_EQ(test, err, -EINVAL);
+ /* CONTIGUOUS + RANGE should succeed via the range-aware try_harder */
+ KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, mm_size,
size,
+ SZ_4K, &blocks,
+
GPU_BUDDY_CONTIGUOUS_ALLOCATION |
+
GPU_BUDDY_RANGE_ALLOCATION),
+ "range contiguous alloc hit an error
size=%llu\n", size);
+ gpu_buddy_free_list(&mm, &blocks, 0);
gpu_buddy_fini(&mm);
}
@@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = {
KUNIT_CASE(gpu_test_buddy_alloc_pessimistic),
KUNIT_CASE(gpu_test_buddy_alloc_pathological),
KUNIT_CASE(gpu_test_buddy_alloc_contiguous),
+ KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous),
KUNIT_CASE(gpu_test_buddy_alloc_clear),
KUNIT_CASE(gpu_test_buddy_alloc_range),
KUNIT_CASE(gpu_test_buddy_alloc_range_bias),
--
2.43.0