https://github.com/skc7 created https://github.com/llvm/llvm-project/pull/220207
**Summary:** - The AMDGPU classifier can pass or return an aggregate directly in registers and can pack a 33-64 bit aggregate into a [2 x i32] register pair. The CallConvLowering bridge handled neither, so such signatures hit the NYI path. Handle both in convertABIArgInfo/abiTypeToCIR. **Changes:** - A Direct classification with a null coerce type is now always a pass-through, for aggregates as well as scalars. Previously only scalars were passed through. An aggregate fell into the coercion path and reported NYI. - `abiTypeToCIR` gains an `ArrayType` case, mapping an `llvm::abi::ArrayType` coerce to a `cir::ArrayType` so the AMDGPU [2 x i32] pack is representable. - Added cir-opt coverage for 64-bit struct coerced to [2 x i32] and larger struct passed directly in registers. - Dropped the `-fno-clangir-call-conv-lowering` opt-out from `amdgcn-buffer-rsrc-type.hip`. Assisted by: Claude Opus 4.8 >From 6a65e58b022eaccf03f3a33c1194d00342927d2f Mon Sep 17 00:00:00 2001 From: skc7 <[email protected]> Date: Tue, 1 Sep 2026 15:01:27 +0530 Subject: [PATCH] [CIR] Support direct-in-registers aggregates and array coercions in call-conv lowering --- .../Transforms/CallConvLoweringPass.cpp | 17 +++++++++++--- .../CodeGenHIP/amdgcn-buffer-rsrc-type.hip | 7 ++---- .../abi-lowering/amdgpu-scalars.cir | 22 +++++++++++++++++++ 3 files changed, 38 insertions(+), 8 deletions(-) diff --git a/clang/lib/CIR/Dialect/Transforms/CallConvLoweringPass.cpp b/clang/lib/CIR/Dialect/Transforms/CallConvLoweringPass.cpp index c1c69947d55c5..c808810391a8d 100644 --- a/clang/lib/CIR/Dialect/Transforms/CallConvLoweringPass.cpp +++ b/clang/lib/CIR/Dialect/Transforms/CallConvLoweringPass.cpp @@ -297,6 +297,12 @@ static mlir::Type abiTypeToCIR(const llvm::abi::Type *ty, MLIRContext *ctx) { return cir::VectorType::get(elemCIR, vecTy->getNumElements().getFixedValue()); }) + .Case([&](const llvm::abi::ArrayType *arrTy) -> mlir::Type { + mlir::Type elemCIR = abiTypeToCIR(arrTy->getElementType(), ctx); + if (!elemCIR) + return nullptr; + return cir::ArrayType::get(elemCIR, arrTy->getNumElements()); + }) .Case([&](const llvm::abi::RecordType *recTy) -> mlir::Type { SmallVector<mlir::Type> fieldTypes; fieldTypes.reserve(recTy->getFields().size()); @@ -455,9 +461,11 @@ static const llvm::abi::Type *mapCIRType(mlir::Type type, /// split into a tuple of them, and a scalar the classifier widens to fill its /// eightbyte. getDirect keeps canFlatten set so the rewriter can split a /// multi-field coerced struct into individual wire arguments. Any other scalar -/// passes in its natural CIR type, which a null coercion denotes. A coercion -/// this bridge cannot represent yields std::nullopt so the caller reports NYI -/// rather than silently passing the value unchanged. +/// passes in its natural CIR type, which a null coercion denotes. An aggregate +/// with a null coercion is likewise passed directly in registers, which the +/// target backend splits. A coercion this bridge cannot represent yields +/// std::nullopt so the caller reports NYI rather than silently passing the +/// value unchanged. /// /// Extend: bool or a sub-register integer needs a signext/zeroext attribute. /// The x86_64 classifier (llvm/lib/ABI/Targets/X86.cpp) only returns Extend @@ -480,6 +488,9 @@ convertABIArgInfo(const llvm::abi::ArgInfo &info, MLIRContext *ctx, // The classifier names a coerce type even where it matches the natural // type, so a non-null coerce does not by itself mean a rewrite is needed. const llvm::abi::Type *coerceAbi = info.getCoerceToType(); + // A null coerce is always a pass-through. + if (!coerceAbi) + return ArgClassification::getDirect(nullptr); bool isAggregate = isa_and_present<cir::RecordType, cir::ArrayType>(origTy); // For a _Complex or a vector the classifier's coerce is only sometimes the // natural type, so it has to be read rather than assumed. diff --git a/clang/test/CIR/CodeGenHIP/amdgcn-buffer-rsrc-type.hip b/clang/test/CIR/CodeGenHIP/amdgcn-buffer-rsrc-type.hip index 0045a731c579f..dbc0b36095d10 100644 --- a/clang/test/CIR/CodeGenHIP/amdgcn-buffer-rsrc-type.hip +++ b/clang/test/CIR/CodeGenHIP/amdgcn-buffer-rsrc-type.hip @@ -2,15 +2,12 @@ // REQUIRES: amdgpu-registered-target -// TODO(cir): drop -fno-clangir-call-conv-lowering once CallConvLowering -// supports the AMDGPU direct-in-registers aggregate return of a buffer -// resource struct. // RUN: %clang_cc1 -triple amdgpu11.00-amd-amdhsa -x hip -std=c++11 -fclangir \ -// RUN: -fcuda-is-device -fno-clangir-call-conv-lowering -emit-cir %s -o %t.cir +// RUN: -fcuda-is-device -emit-cir %s -o %t.cir // RUN: FileCheck --check-prefix=CIR --input-file=%t.cir %s // RUN: %clang_cc1 -triple amdgpu11.00-amd-amdhsa -x hip -std=c++11 -fclangir \ -// RUN: -fcuda-is-device -fno-clangir-call-conv-lowering -emit-llvm %s -o %t-cir.ll +// RUN: -fcuda-is-device -emit-llvm %s -o %t-cir.ll // RUN: FileCheck --check-prefix=LLVM --input-file=%t-cir.ll %s // RUN: %clang_cc1 -triple amdgpu11.00-amd-amdhsa -x hip -std=c++11 \ diff --git a/clang/test/CIR/Transforms/abi-lowering/amdgpu-scalars.cir b/clang/test/CIR/Transforms/abi-lowering/amdgpu-scalars.cir index 5b9049e119757..60e407ca407dd 100644 --- a/clang/test/CIR/Transforms/abi-lowering/amdgpu-scalars.cir +++ b/clang/test/CIR/Transforms/abi-lowering/amdgpu-scalars.cir @@ -4,10 +4,13 @@ !s16i = !cir.int<s, 16> !u16i = !cir.int<u, 16> !s32i = !cir.int<s, 32> +!u32i = !cir.int<u, 32> !s64i = !cir.int<s, 64> !rec_E0 = !cir.struct<"E0" {}> !rec_Pair16 = !cir.struct<"Pair16" {data !s16i, data !s16i}> +!rec_Pair32 = !cir.struct<"Pair32" {data !s32i, data !s32i}> +!rec_Triple = !cir.struct<"Triple" {data !s32i, data !s32i, data !s32i}> module attributes { dlti.dl_spec = #dlti.dl_spec< @@ -95,6 +98,25 @@ module attributes { // CHECK: cir.func{{.*}} @takes_small(%{{.*}}: !u32i) + // A 33-64 bit aggregate is packed into a pair of i32 registers, coerced on + // the wire to a [2 x i32] array. + cir.func @takes_pair32(%arg0: !rec_Pair32) { + %0 = cir.alloca "p" align(4) : !cir.ptr<!rec_Pair32> + cir.store %arg0, %0 : !rec_Pair32, !cir.ptr<!rec_Pair32> + cir.return + } + + // CHECK: cir.func{{.*}} @takes_pair32(%{{.*}}: !cir.array<!u32i x 2>) + + // A larger aggregate that still fits the register budget is passed directly + // in registers. The signature is left unchanged and the target backend will + // split it into registers. + cir.func @takes_triple(%arg0: !rec_Triple) { + cir.return + } + + // CHECK: cir.func{{.*}} @takes_triple(%arg0: !rec_Triple) + // A kernel scalar argument stays Direct: the AMDGPU_KERNEL convention is // routed to the kernel classifier and the signature is unchanged. cir.func @kernel_scalar(%arg0: !s32i) cc(amdgpu_kernel) { _______________________________________________ llvm-branch-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/llvm-branch-commits
