From: Arunpravin Paneer Selvam <[email protected]>

A range + contiguous allocation (e.g. a scanout FB confined to the
CPU-visible VRAM aperture) rounds its size up to a power of two and
requires a naturally aligned free block of that size; on a fragmented
aperture no such aligned block may exist even though enough contiguous
space is free, so the allocation fails with -ENOSPC.

The non-range contiguous path already recovers from this via
__alloc_contig_try_harder(), which stitches an exact-size span from
smaller adjacent blocks, but that fallback was unreachable once
GPU_BUDDY_RANGE_ALLOCATION was set. Give __alloc_contig_try_harder()
a [range_start, range_end) window and route the range + contiguous
case through it: each candidate placement is confined to the window
and aligned to min_block_size. Non-range callers pass [0, mm->size),
where the guards are no-ops, so the existing behaviour is unchanged.
A KUnit regression test covers the fragmentation pattern.

Resolves the igt@kms_plane@plane-panning-bottom-right@pipe-a/pipe-b
regression.

Fixes: 1ad5e807f716 ("gpu/buddy: replace dual-tree/force_merge with decoupled 
dirty tracker")
Assisted-by: Claude:claude-opus-4-8
Cc: Matthew Auld <[email protected]>
Cc: Christian König <[email protected]>
Signed-off-by: Arunpravin Paneer Selvam <[email protected]>
---
 drivers/gpu/buddy.c                | 88 ++++++++++++++++--------------
 drivers/gpu/tests/gpu_buddy_test.c | 83 ++++++++++++++++++++++++++--
 2 files changed, 126 insertions(+), 45 deletions(-)

diff --git a/drivers/gpu/buddy.c b/drivers/gpu/buddy.c
index e5c9e21cd077..08e82378a9f0 100644
--- a/drivers/gpu/buddy.c
+++ b/drivers/gpu/buddy.c
@@ -1735,26 +1735,14 @@ static int __gpu_buddy_alloc_range(struct gpu_buddy *mm,
                             blocks, total_allocated_on_err);
 }
 
-static int __alloc_contig_aligned_retry(struct gpu_buddy *mm,
-                                       u64 unaligned_offset,
-                                       u64 size,
-                                       u64 min_block_size,
-                                       unsigned long flags,
-                                       struct list_head *blocks)
-{
-       u64 aligned_offset = round_down(unaligned_offset, min_block_size);
-
-       return __gpu_buddy_alloc_range(mm, aligned_offset, size, flags,
-                                      NULL, blocks);
-}
-
 static int __alloc_contig_try_harder(struct gpu_buddy *mm,
+                                    u64 range_start, u64 range_end,
                                     u64 size,
                                     u64 min_block_size,
                                     unsigned long flags,
                                     struct list_head *blocks)
 {
-       u64 rhs_offset, lhs_offset, filled;
+       u64 rhs_offset, lhs_offset, filled, aligned;
        struct gpu_buddy_block *block;
        struct rb_root *root;
        struct rb_node *iter;
@@ -1784,20 +1772,24 @@ static int __alloc_contig_try_harder(struct gpu_buddy 
*mm,
                                               flags, &filled, blocks);
                if (err && err != -ENOSPC)
                        return err;
-               if (!err && IS_ALIGNED(rhs_offset, min_block_size))
+               if (!err && IS_ALIGNED(rhs_offset, min_block_size) &&
+                   rhs_offset >= range_start && rhs_offset + size <= range_end)
                        return 0;
                if (!err) {
                        /* Allocate the unaligned RHS offset using round_down */
                        gpu_buddy_free_list_internal(mm, blocks);
-                       err = __alloc_contig_aligned_retry(mm, rhs_offset,
-                                                          size,
-                                                          min_block_size,
-                                                          flags, blocks);
-                       if (!err)
-                               return 0;
-                       if (err != -ENOSPC) {
-                               gpu_buddy_free_list_internal(mm, blocks);
-                               return err;
+
+                       aligned = round_down(rhs_offset, min_block_size);
+                       if (aligned >= range_start &&
+                           aligned + size <= range_end) {
+                               err = __gpu_buddy_alloc_range(mm, aligned, size,
+                                                             flags, NULL, 
blocks);
+                               if (!err)
+                                       return 0;
+                               if (err != -ENOSPC) {
+                                       gpu_buddy_free_list_internal(mm, 
blocks);
+                                       return err;
+                               }
                        }
                        goto next;
                }
@@ -1809,15 +1801,17 @@ static int __alloc_contig_try_harder(struct gpu_buddy 
*mm,
 
                /* Allocate the unaligned LHS offset using round_down */
                gpu_buddy_free_list_internal(mm, blocks);
-               err = __alloc_contig_aligned_retry(mm, lhs_offset,
-                                                  size,
-                                                  min_block_size,
-                                                  flags, blocks);
-               if (!err)
-                       return 0;
-               if (err != -ENOSPC) {
-                       gpu_buddy_free_list_internal(mm, blocks);
-                       return err;
+
+               aligned = round_down(lhs_offset, min_block_size);
+               if (aligned >= range_start && aligned + size <= range_end) {
+                       err = __gpu_buddy_alloc_range(mm, aligned, size,
+                                                     flags, NULL, blocks);
+                       if (!err)
+                               return 0;
+                       if (err != -ENOSPC) {
+                               gpu_buddy_free_list_internal(mm, blocks);
+                               return err;
+                       }
                }
 next:
                gpu_buddy_free_list_internal(mm, blocks);
@@ -2071,11 +2065,18 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
        min_order = ilog2(min_block_size) - ilog2(mm->chunk_size);
 
        if (order > mm->max_order || size > mm->size) {
-               if ((flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) &&
-                   !(flags & GPU_BUDDY_RANGE_ALLOCATION))
-                       return __alloc_contig_try_harder(mm, original_size,
+               if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
+                       u64 range_start, range_end;
+
+                       range_start = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? 
start : 0;
+                       range_end = (flags & GPU_BUDDY_RANGE_ALLOCATION) ? end 
: mm->size;
+
+                       return __alloc_contig_try_harder(mm, range_start,
+                                                        range_end,
+                                                        original_size,
                                                         original_min_size,
                                                         flags, blocks);
+               }
 
                return -EINVAL;
        }
@@ -2108,9 +2109,14 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
                         * Try contiguous block allocation through
                         * try harder method.
                         */
-                       if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION &&
-                           !(flags & GPU_BUDDY_RANGE_ALLOCATION)) {
-                               err = __alloc_contig_try_harder(mm,
+                       if (flags & GPU_BUDDY_CONTIGUOUS_ALLOCATION) {
+                               u64 range_start, range_end;
+
+                               range_start = (flags & 
GPU_BUDDY_RANGE_ALLOCATION) ? start : 0;
+                               range_end = (flags & 
GPU_BUDDY_RANGE_ALLOCATION) ? end : mm->size;
+
+                               err = __alloc_contig_try_harder(mm, range_start,
+                                                               range_end,
                                                                original_size,
                                                                
original_min_size,
                                                                flags,
@@ -2118,9 +2124,9 @@ int gpu_buddy_alloc_blocks(struct gpu_buddy *mm,
                                if (!err)
                                        return 0;
                                if (err != -ENOSPC)
-                                       return err;
-                               goto err_free;
+                                       goto err_free;
                        }
+
                        err = -ENOSPC;
                        goto err_free;
                } while (1);
diff --git a/drivers/gpu/tests/gpu_buddy_test.c 
b/drivers/gpu/tests/gpu_buddy_test.c
index b75d32ca6ca0..2c445870b808 100644
--- a/drivers/gpu/tests/gpu_buddy_test.c
+++ b/drivers/gpu/tests/gpu_buddy_test.c
@@ -1251,6 +1251,77 @@ static void gpu_test_buddy_alloc_contiguous(struct kunit 
*test)
        gpu_buddy_fini(&mm);
 }
 
+static void gpu_test_buddy_alloc_range_contiguous(struct kunit *test)
+{
+       const unsigned long ps = SZ_4K, mm_size = 16 * ps;
+       const unsigned long range_end = 8 * ps;
+       struct gpu_buddy_block *block, *prev;
+       LIST_HEAD(allocated);
+       struct gpu_buddy mm;
+       LIST_HEAD(pin_lo);
+       LIST_HEAD(pin_hi);
+       u64 total;
+
+       KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_init(&mm, mm_size, ps),
+                              "buddy_init failed\n");
+
+       /*
+        * Idea is to confine the test to the sub-range [0, 32K), which a 12K
+        * contiguous request (rounded up to 16K) splits into two naturally
+        * aligned 16K slots: [0, 16K) and [16K, 32K). We pin the first 4K page
+        * of each slot ([0, 4K) and [16K, 20K)) so that neither aligned slot
+        * can satisfy the rounded-up 16K allocation, yet the freed remainder
+        * still leaves a contiguous 12K hole at offset 4K, which is 
page-aligned
+        * but not 16K-aligned. A 12K contiguous+range allocation must therefore
+        * fall back to stitching that span instead of returning -ENOSPC.
+        */
+       KUNIT_ASSERT_FALSE_MSG(test,
+                              gpu_buddy_alloc_blocks(&mm, 0, ps, ps, ps,
+                                                     &pin_lo, 0),
+                              "failed to pin low page\n");
+       KUNIT_ASSERT_FALSE_MSG(test,
+                              gpu_buddy_alloc_blocks(&mm, 4 * ps, 5 * ps, ps,
+                                                     ps, &pin_hi, 0),
+                              "failed to pin high page\n");
+
+       /* No aligned 16K block is free; the range-aware fallback must stitch
+        * the unaligned [ps, 4*ps) hole instead of returning -ENOSPC.
+        */
+       KUNIT_ASSERT_FALSE_MSG(test,
+                              gpu_buddy_alloc_blocks(&mm, 0, range_end, 3 * ps,
+                                                     ps, &allocated,
+                                                     
GPU_BUDDY_CONTIGUOUS_ALLOCATION |
+                                                     
GPU_BUDDY_RANGE_ALLOCATION),
+                              "range-restricted contiguous alloc failed\n");
+
+       /* The result must be exactly 3*ps, contiguous, and inside the range. */
+       total = 0;
+       prev = NULL;
+       list_for_each_entry(block, &allocated, link) {
+               u64 offset = gpu_buddy_block_offset(block);
+               u64 bsize = gpu_buddy_block_size(&mm, block);
+
+               KUNIT_EXPECT_TRUE_MSG(test, offset + bsize <= range_end,
+                                     "block [%llx, %llx) outside range\n",
+                                     offset, offset + bsize);
+               if (prev)
+                       KUNIT_EXPECT_EQ_MSG(test,
+                                           gpu_buddy_block_offset(prev) +
+                                           gpu_buddy_block_size(&mm, prev),
+                                           offset,
+                                           "block at %llx not contiguous\n",
+                                           offset);
+               prev = block;
+               total += bsize;
+       }
+       KUNIT_EXPECT_EQ(test, total, 3 * ps);
+
+       gpu_buddy_free_list(&mm, &allocated, 0);
+       gpu_buddy_free_list(&mm, &pin_lo, 0);
+       gpu_buddy_free_list(&mm, &pin_hi, 0);
+       gpu_buddy_fini(&mm);
+}
+
 static void gpu_test_buddy_alloc_pathological(struct kunit *test)
 {
        u64 mm_size, size, start = 0;
@@ -1534,10 +1605,13 @@ static void 
gpu_test_buddy_alloc_exceeds_max_order(struct kunit *test)
                                     GPU_BUDDY_RANGE_ALLOCATION);
        KUNIT_EXPECT_EQ(test, err, -EINVAL);
 
-       /* CONTIGUOUS + RANGE should return -EINVAL (no try_harder for RANGE) */
-       err = gpu_buddy_alloc_blocks(&mm, 0, mm_size, size, SZ_4K, &blocks,
-                                    GPU_BUDDY_CONTIGUOUS_ALLOCATION | 
GPU_BUDDY_RANGE_ALLOCATION);
-       KUNIT_EXPECT_EQ(test, err, -EINVAL);
+       /* CONTIGUOUS + RANGE should succeed via the range-aware try_harder */
+       KUNIT_ASSERT_FALSE_MSG(test, gpu_buddy_alloc_blocks(&mm, 0, mm_size, 
size,
+                                                           SZ_4K, &blocks,
+                                                           
GPU_BUDDY_CONTIGUOUS_ALLOCATION |
+                                                           
GPU_BUDDY_RANGE_ALLOCATION),
+                              "range contiguous alloc hit an error 
size=%llu\n", size);
+       gpu_buddy_free_list(&mm, &blocks, 0);
 
        gpu_buddy_fini(&mm);
 }
@@ -1603,6 +1677,7 @@ static struct kunit_case gpu_buddy_tests[] = {
        KUNIT_CASE(gpu_test_buddy_alloc_pessimistic),
        KUNIT_CASE(gpu_test_buddy_alloc_pathological),
        KUNIT_CASE(gpu_test_buddy_alloc_contiguous),
+       KUNIT_CASE(gpu_test_buddy_alloc_range_contiguous),
        KUNIT_CASE(gpu_test_buddy_alloc_clear),
        KUNIT_CASE(gpu_test_buddy_alloc_range),
        KUNIT_CASE(gpu_test_buddy_alloc_range_bias),
-- 
2.43.0

Reply via email to