https://github.com/farzonl updated https://github.com/llvm/llvm-project/pull/227049
>From baebb44f93ba07c4b35d856568810eba5f3f0c0d Mon Sep 17 00:00:00 2001 From: Farzon Lotfi <[email protected]> Date: Tue, 22 Sep 2026 17:01:45 -0400 Subject: [PATCH 1/2] [HLSL] Fold inverse matrix transposes When the operand is an unused llvm.matrix.transpose with inverse dimensions, return its original operand and erase the redundant intrinsic. This handles both explicit transpose(transpose(M)) expressions and transposes that cancel row-major load normalization. --- clang/lib/CodeGen/CGHLSLBuiltins.cpp | 12 +++++++ .../test/CodeGenHLSL/builtins/transpose.hlsl | 34 +++++++++++++------ 2 files changed, 35 insertions(+), 11 deletions(-) diff --git a/clang/lib/CodeGen/CGHLSLBuiltins.cpp b/clang/lib/CodeGen/CGHLSLBuiltins.cpp index eec95f64297e4..19c620a6e01e8 100644 --- a/clang/lib/CodeGen/CGHLSLBuiltins.cpp +++ b/clang/lib/CodeGen/CGHLSLBuiltins.cpp @@ -1316,6 +1316,18 @@ Value *CodeGenFunction::EmitHLSLBuiltinExpr(unsigned BuiltinID, auto *MatTy = E->getArg(0)->getType()->castAs<ConstantMatrixType>(); unsigned Rows = MatTy->getNumRows(); unsigned Cols = MatTy->getNumColumns(); + if (auto *Transpose = dyn_cast<CallInst>(Op0); + Transpose && + Transpose->getIntrinsicID() == Intrinsic::matrix_transpose && + Transpose->use_empty() && + cast<ConstantInt>(Transpose->getArgOperand(1))->getZExtValue() == + Cols && + cast<ConstantInt>(Transpose->getArgOperand(2))->getZExtValue() == + Rows) { + Value *Result = Transpose->getArgOperand(0); + Transpose->eraseFromParent(); + return Result; + } llvm::MatrixBuilder MB(Builder); return MB.CreateMatrixTranspose(Op0, Rows, Cols); } diff --git a/clang/test/CodeGenHLSL/builtins/transpose.hlsl b/clang/test/CodeGenHLSL/builtins/transpose.hlsl index 96198d8b08e9d..09fe32e01e0be 100644 --- a/clang/test/CodeGenHLSL/builtins/transpose.hlsl +++ b/clang/test/CodeGenHLSL/builtins/transpose.hlsl @@ -10,8 +10,7 @@ // CHECK: store <6 x i32> [[A_EXT]], ptr [[A_ADDR]], align 4 // CHECK: [[A:%.*]] = load <6 x i32>, ptr [[A_ADDR]], align 4 // COLMAJOR: [[TRANS:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[A]], i32 2, i32 3) -// ROWMAJOR: [[NORMALIZED:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[A]], i32 3, i32 2) -// ROWMAJOR: [[TRANS:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[NORMALIZED]], i32 2, i32 3) +// ROWMAJOR-NOT: call {{.*}} @llvm.matrix.transpose bool3x2 test_transpose_bool2x3(bool2x3 a) { return transpose(a); } @@ -22,9 +21,9 @@ bool3x2 test_transpose_bool2x3(bool2x3 a) { // CHECK: store <12 x i32> %{{.*}}, ptr [[A_ADDR]], align 4 // CHECK: [[A:%.*]] = load <12 x i32>, ptr [[A_ADDR]], align 4 // COLMAJOR: [[TRANS:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[A]], i32 4, i32 3) -// ROWMAJOR: [[NORMALIZED:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[A]], i32 3, i32 4) -// ROWMAJOR: [[TRANS:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[NORMALIZED]], i32 4, i32 3) -// CHECK: ret <12 x i32> [[TRANS]] +// ROWMAJOR-NOT: call {{.*}} @llvm.matrix.transpose +// COLMAJOR: ret <12 x i32> [[TRANS]] +// ROWMAJOR: ret <12 x i32> [[A]] int3x4 test_transpose_int4x3(int4x3 a) { return transpose(a); } @@ -34,9 +33,9 @@ int3x4 test_transpose_int4x3(int4x3 a) { // CHECK: store <16 x float> %{{.*}}, ptr [[A_ADDR]], align 4 // CHECK: [[A:%.*]] = load <16 x float>, ptr [[A_ADDR]], align 4 // COLMAJOR: [[TRANS:%.*]] = call {{.*}}<16 x float> @llvm.matrix.transpose.v16f32(<16 x float> [[A]], i32 4, i32 4) -// ROWMAJOR: [[NORMALIZED:%.*]] = call {{.*}}<16 x float> @llvm.matrix.transpose.v16f32(<16 x float> [[A]], i32 4, i32 4) -// ROWMAJOR: [[TRANS:%.*]] = call {{.*}}<16 x float> @llvm.matrix.transpose.v16f32(<16 x float> [[NORMALIZED]], i32 4, i32 4) -// CHECK: ret <16 x float> [[TRANS]] +// ROWMAJOR-NOT: call {{.*}} @llvm.matrix.transpose +// COLMAJOR: ret <16 x float> [[TRANS]] +// ROWMAJOR: ret <16 x float> [[A]] float4x4 test_transpose_float4x4(float4x4 a) { return transpose(a); } @@ -47,9 +46,22 @@ float4x4 test_transpose_float4x4(float4x4 a) { // CHECK: store <4 x double> %{{.*}}, ptr [[A_ADDR]], align 8 // CHECK: [[A:%.*]] = load <4 x double>, ptr [[A_ADDR]], align 8 // COLMAJOR: [[TRANS:%.*]] = call {{.*}}<4 x double> @llvm.matrix.transpose.v4f64(<4 x double> [[A]], i32 1, i32 4) -// ROWMAJOR: [[NORMALIZED:%.*]] = call {{.*}}<4 x double> @llvm.matrix.transpose.v4f64(<4 x double> [[A]], i32 4, i32 1) -// ROWMAJOR: [[TRANS:%.*]] = call {{.*}}<4 x double> @llvm.matrix.transpose.v4f64(<4 x double> [[NORMALIZED]], i32 1, i32 4) -// CHECK: ret <4 x double> [[TRANS]] +// ROWMAJOR-NOT: call {{.*}} @llvm.matrix.transpose +// COLMAJOR: ret <4 x double> [[TRANS]] +// ROWMAJOR: ret <4 x double> [[A]] double4x1 test_transpose_double1x4(double1x4 a) { return transpose(a); } + +// CHECK-LABEL: define {{.*}}test_double_transpose_float2x3 +// COLMAJOR: [[A_ADDR:%.*]] = alloca [3 x <2 x float>], align 4 +// ROWMAJOR: [[A_ADDR:%.*]] = alloca [2 x <3 x float>], align 4 +// CHECK: store <6 x float> %{{.*}}, ptr [[A_ADDR]], align 4 +// CHECK: [[A:%.*]] = load <6 x float>, ptr [[A_ADDR]], align 4 +// COLMAJOR-NOT: call {{.*}} @llvm.matrix.transpose +// ROWMAJOR: [[NORMALIZED:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> [[A]], i32 3, i32 2) +// COLMAJOR: ret <6 x float> [[A]] +// ROWMAJOR: ret <6 x float> [[NORMALIZED]] +float2x3 test_double_transpose_float2x3(float2x3 a) { + return transpose(transpose(a)); +} >From 6e5e31ffcd0f698e8f04430e78a7ffdb26d44a70 Mon Sep 17 00:00:00 2001 From: Farzon Lotfi <[email protected]> Date: Mon, 28 Sep 2026 13:48:53 -0400 Subject: [PATCH 2/2] update matrix-layout-attr-overrides-default.hlsl after rebase with main --- .../matrix-layout-attr-overrides-default.hlsl | 25 +++++++++++++------ 1 file changed, 17 insertions(+), 8 deletions(-) diff --git a/clang/test/CodeGenHLSL/matrix-layout-attr-overrides-default.hlsl b/clang/test/CodeGenHLSL/matrix-layout-attr-overrides-default.hlsl index a04cf58b45b7c..d683e6310e807 100644 --- a/clang/test/CodeGenHLSL/matrix-layout-attr-overrides-default.hlsl +++ b/clang/test/CodeGenHLSL/matrix-layout-attr-overrides-default.hlsl @@ -116,22 +116,28 @@ export row_major float2x2 mat_mat_dst_rm(column_major float2x3 a, column_major f // CHECK: ret <4 x float> [[MUL]] -// Transpose operates on the canonical column-major value after the load. +// The transpose cancels the row-major load normalization. export column_major float3x2 transpose_rm_to_cm(row_major float2x3 m) { return transpose(m); } // CHECK-LABEL: define {{.*}} <6 x float> @_Z18transpose_rm_to_cmu11matrix_typeILm2ELm3EfE -// CHECK: [[RM:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 3, i32 2) -// CHECK: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> [[RM]], i32 2, i32 3) +// CHECK: [[TO_MEMORY:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 2, i32 3) +// CHECK: store <6 x float> [[TO_MEMORY]], ptr %{{.*}} +// CHECK: [[FROM_MEMORY:%.*]] = load <6 x float>, ptr %{{.*}} +// CHECK-NOT: @llvm.matrix.transpose +// CHECK: ret <6 x float> [[FROM_MEMORY]] // Return layout metadata does not change the canonical value representation. export row_major float3x2 transpose_cm_to_rm(column_major float2x3 m) { return transpose(m); } // CHECK-LABEL: define {{.*}} <6 x float> @_Z18transpose_cm_to_rmu11matrix_typeILm2ELm3EfE // CHECK: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 2, i32 3) -// Row-major source -> row-major destination: real transpose, dims swapped. +// Row-major source -> row-major destination: the transpose cancels load normalization. export row_major float3x2 transpose_rm_to_rm(row_major float2x3 m) { return transpose(m); } // CHECK-LABEL: define {{.*}} <6 x float> @_Z18transpose_rm_to_rmu11matrix_typeILm2ELm3EfE -// CHECK: [[RM:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 3, i32 2) -// CHECK: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> [[RM]], i32 2, i32 3) +// CHECK: [[TO_MEMORY:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 2, i32 3) +// CHECK: store <6 x float> [[TO_MEMORY]], ptr %{{.*}} +// CHECK: [[FROM_MEMORY:%.*]] = load <6 x float>, ptr %{{.*}} +// CHECK-NOT: @llvm.matrix.transpose +// CHECK: ret <6 x float> [[FROM_MEMORY]] // Column-major source -> column-major destination: real transpose, natural dims. export column_major float3x2 transpose_cm_to_cm(column_major float2x3 m) { return transpose(m); } @@ -141,8 +147,11 @@ export column_major float3x2 transpose_cm_to_cm(column_major float2x3 m) { retur // The TU memory-layout default does not affect matrix prvalues. export float3x2 transpose_rm(row_major float2x3 m) { return transpose(m); } // CHECK-LABEL: define {{.*}} <6 x float> @_Z12transpose_rmu11matrix_typeILm2ELm3EfE -// CHECK: [[RM:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 3, i32 2) -// CHECK: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> [[RM]], i32 2, i32 3) +// CHECK: [[TO_MEMORY:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 2, i32 3) +// CHECK: store <6 x float> [[TO_MEMORY]], ptr %{{.*}} +// CHECK: [[FROM_MEMORY:%.*]] = load <6 x float>, ptr %{{.*}} +// CHECK-NOT: @llvm.matrix.transpose +// CHECK: ret <6 x float> [[FROM_MEMORY]] export float3x2 transpose_cm(column_major float2x3 m) { return transpose(m); } _______________________________________________ cfe-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits
