https://github.com/E00N777 created 
https://github.com/llvm/llvm-project/pull/226382

## summary

part of : https://github.com/llvm/llvm-project/issues/185382

Lower all intrinsics in : 
https://arm-software.github.io/acle/neon_intrinsics/advsimd.html#vector-shift-right-and-narrow



>From fb86e7d1f4c247fd5aca7fd5036027d934d74041 Mon Sep 17 00:00:00 2001
From: E00N777 <[email protected]>
Date: Fri, 25 Sep 2026 14:53:07 +0800
Subject: [PATCH] [CIR][AArch64] Lower NEON vshrn* intrinsics

---
 .../lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp  |  16 +-
 clang/test/CodeGen/AArch64/neon-intrinsics.c  | 162 ----------------
 clang/test/CodeGen/AArch64/neon/intrinsics.c  | 173 ++++++++++++++++++
 3 files changed, 184 insertions(+), 167 deletions(-)

diff --git a/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp 
b/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp
index 83cd4255a5bb4e..ff4535564d1210 100644
--- a/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp
+++ b/clang/lib/CIR/CodeGen/CIRGenBuiltinAArch64.cpp
@@ -1287,11 +1287,17 @@ static mlir::Value emitCommonNeonBuiltinExpr(
     return emitCommonNeonShift(builder, loc, vTy, extended, ops[1],
                                /*shiftLeft=*/true);
   }
-  case NEON::BI__builtin_neon_vshrn_n_v:
-    cgf.cgm.errorNYI(expr->getSourceRange(),
-                     std::string("unimplemented AArch64 builtin call: ") +
-                         ctx.BuiltinInfo.getName(builtinID));
-    return mlir::Value{};
+  case NEON::BI__builtin_neon_vshrn_n_v: {
+    CIRGenBuilderTy &builder = cgf.getBuilder();
+    cir::VectorType wideVecTy =
+        builder.getExtendedOrTruncatedElementVectorType(vTy,
+                                                        /*isExtended=*/true,
+                                                        /*isSigned=*/!usgn);
+    mlir::Value src = builder.createBitcast(ops[0], wideVecTy);
+    mlir::Value shifted = emitCommonNeonShift(builder, loc, wideVecTy, src,
+                                              ops[1], /*shiftLeft=*/false);
+    return builder.createIntCast(shifted, vTy);
+  }
   case NEON::BI__builtin_neon_vshr_n_v:
   case NEON::BI__builtin_neon_vshrq_n_v:
     return emitNeonRShiftImm(cgf, ops[0], ops[1], vTy, isUnsigned, loc);
diff --git a/clang/test/CodeGen/AArch64/neon-intrinsics.c 
b/clang/test/CodeGen/AArch64/neon-intrinsics.c
index 2098abb0978893..9934c13d7e96f5 100644
--- a/clang/test/CodeGen/AArch64/neon-intrinsics.c
+++ b/clang/test/CodeGen/AArch64/neon-intrinsics.c
@@ -3242,168 +3242,6 @@ float64x2_t test_vmulxq_f64(float64x2_t a, float64x2_t 
b) {
   return vmulxq_f64(a, b);
 }
 
-// CHECK-LABEL: define dso_local <8 x i8> @test_vshrn_n_s16(
-// CHECK-SAME: <8 x i16> noundef [[A:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    [[TMP0:%.*]] = bitcast <8 x i16> [[A]] to <16 x i8>
-// CHECK-NEXT:    [[TMP1:%.*]] = bitcast <16 x i8> [[TMP0]] to <8 x i16>
-// CHECK-NEXT:    [[TMP2:%.*]] = ashr <8 x i16> [[TMP1]], splat (i16 3)
-// CHECK-NEXT:    [[VSHRN_N:%.*]] = trunc <8 x i16> [[TMP2]] to <8 x i8>
-// CHECK-NEXT:    ret <8 x i8> [[VSHRN_N]]
-//
-int8x8_t test_vshrn_n_s16(int16x8_t a) {
-  return vshrn_n_s16(a, 3);
-}
-
-// CHECK-LABEL: define dso_local <4 x i16> @test_vshrn_n_s32(
-// CHECK-SAME: <4 x i32> noundef [[A:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    [[TMP0:%.*]] = bitcast <4 x i32> [[A]] to <16 x i8>
-// CHECK-NEXT:    [[TMP1:%.*]] = bitcast <16 x i8> [[TMP0]] to <4 x i32>
-// CHECK-NEXT:    [[TMP2:%.*]] = ashr <4 x i32> [[TMP1]], splat (i32 9)
-// CHECK-NEXT:    [[VSHRN_N:%.*]] = trunc <4 x i32> [[TMP2]] to <4 x i16>
-// CHECK-NEXT:    ret <4 x i16> [[VSHRN_N]]
-//
-int16x4_t test_vshrn_n_s32(int32x4_t a) {
-  return vshrn_n_s32(a, 9);
-}
-
-// CHECK-LABEL: define dso_local <2 x i32> @test_vshrn_n_s64(
-// CHECK-SAME: <2 x i64> noundef [[A:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    [[TMP0:%.*]] = bitcast <2 x i64> [[A]] to <16 x i8>
-// CHECK-NEXT:    [[TMP1:%.*]] = bitcast <16 x i8> [[TMP0]] to <2 x i64>
-// CHECK-NEXT:    [[TMP2:%.*]] = ashr <2 x i64> [[TMP1]], splat (i64 19)
-// CHECK-NEXT:    [[VSHRN_N:%.*]] = trunc <2 x i64> [[TMP2]] to <2 x i32>
-// CHECK-NEXT:    ret <2 x i32> [[VSHRN_N]]
-//
-int32x2_t test_vshrn_n_s64(int64x2_t a) {
-  return vshrn_n_s64(a, 19);
-}
-
-// CHECK-LABEL: define dso_local <8 x i8> @test_vshrn_n_u16(
-// CHECK-SAME: <8 x i16> noundef [[A:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    [[TMP0:%.*]] = bitcast <8 x i16> [[A]] to <16 x i8>
-// CHECK-NEXT:    [[TMP1:%.*]] = bitcast <16 x i8> [[TMP0]] to <8 x i16>
-// CHECK-NEXT:    [[TMP2:%.*]] = lshr <8 x i16> [[TMP1]], splat (i16 3)
-// CHECK-NEXT:    [[VSHRN_N:%.*]] = trunc <8 x i16> [[TMP2]] to <8 x i8>
-// CHECK-NEXT:    ret <8 x i8> [[VSHRN_N]]
-//
-uint8x8_t test_vshrn_n_u16(uint16x8_t a) {
-  return vshrn_n_u16(a, 3);
-}
-
-// CHECK-LABEL: define dso_local <4 x i16> @test_vshrn_n_u32(
-// CHECK-SAME: <4 x i32> noundef [[A:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    [[TMP0:%.*]] = bitcast <4 x i32> [[A]] to <16 x i8>
-// CHECK-NEXT:    [[TMP1:%.*]] = bitcast <16 x i8> [[TMP0]] to <4 x i32>
-// CHECK-NEXT:    [[TMP2:%.*]] = lshr <4 x i32> [[TMP1]], splat (i32 9)
-// CHECK-NEXT:    [[VSHRN_N:%.*]] = trunc <4 x i32> [[TMP2]] to <4 x i16>
-// CHECK-NEXT:    ret <4 x i16> [[VSHRN_N]]
-//
-uint16x4_t test_vshrn_n_u32(uint32x4_t a) {
-  return vshrn_n_u32(a, 9);
-}
-
-// CHECK-LABEL: define dso_local <2 x i32> @test_vshrn_n_u64(
-// CHECK-SAME: <2 x i64> noundef [[A:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    [[TMP0:%.*]] = bitcast <2 x i64> [[A]] to <16 x i8>
-// CHECK-NEXT:    [[TMP1:%.*]] = bitcast <16 x i8> [[TMP0]] to <2 x i64>
-// CHECK-NEXT:    [[TMP2:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 19)
-// CHECK-NEXT:    [[VSHRN_N:%.*]] = trunc <2 x i64> [[TMP2]] to <2 x i32>
-// CHECK-NEXT:    ret <2 x i32> [[VSHRN_N]]
-//
-uint32x2_t test_vshrn_n_u64(uint64x2_t a) {
-  return vshrn_n_u64(a, 19);
-}
-
-// CHECK-LABEL: define dso_local <16 x i8> @test_vshrn_high_n_s16(
-// CHECK-SAME: <8 x i8> noundef [[A:%.*]], <8 x i16> noundef [[B:%.*]]) 
#[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    [[TMP0:%.*]] = bitcast <8 x i16> [[B]] to <16 x i8>
-// CHECK-NEXT:    [[TMP1:%.*]] = bitcast <16 x i8> [[TMP0]] to <8 x i16>
-// CHECK-NEXT:    [[TMP2:%.*]] = ashr <8 x i16> [[TMP1]], splat (i16 3)
-// CHECK-NEXT:    [[VSHRN_N:%.*]] = trunc <8 x i16> [[TMP2]] to <8 x i8>
-// CHECK-NEXT:    [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[A]], <8 x i8> 
[[VSHRN_N]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 
7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
-// CHECK-NEXT:    ret <16 x i8> [[SHUFFLE_I]]
-//
-int8x16_t test_vshrn_high_n_s16(int8x8_t a, int16x8_t b) {
-  return vshrn_high_n_s16(a, b, 3);
-}
-
-// CHECK-LABEL: define dso_local <8 x i16> @test_vshrn_high_n_s32(
-// CHECK-SAME: <4 x i16> noundef [[A:%.*]], <4 x i32> noundef [[B:%.*]]) 
#[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    [[TMP0:%.*]] = bitcast <4 x i32> [[B]] to <16 x i8>
-// CHECK-NEXT:    [[TMP1:%.*]] = bitcast <16 x i8> [[TMP0]] to <4 x i32>
-// CHECK-NEXT:    [[TMP2:%.*]] = ashr <4 x i32> [[TMP1]], splat (i32 9)
-// CHECK-NEXT:    [[VSHRN_N:%.*]] = trunc <4 x i32> [[TMP2]] to <4 x i16>
-// CHECK-NEXT:    [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[A]], <4 x i16> 
[[VSHRN_N]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-// CHECK-NEXT:    ret <8 x i16> [[SHUFFLE_I]]
-//
-int16x8_t test_vshrn_high_n_s32(int16x4_t a, int32x4_t b) {
-  return vshrn_high_n_s32(a, b, 9);
-}
-
-// CHECK-LABEL: define dso_local <4 x i32> @test_vshrn_high_n_s64(
-// CHECK-SAME: <2 x i32> noundef [[A:%.*]], <2 x i64> noundef [[B:%.*]]) 
#[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    [[TMP0:%.*]] = bitcast <2 x i64> [[B]] to <16 x i8>
-// CHECK-NEXT:    [[TMP1:%.*]] = bitcast <16 x i8> [[TMP0]] to <2 x i64>
-// CHECK-NEXT:    [[TMP2:%.*]] = ashr <2 x i64> [[TMP1]], splat (i64 19)
-// CHECK-NEXT:    [[VSHRN_N:%.*]] = trunc <2 x i64> [[TMP2]] to <2 x i32>
-// CHECK-NEXT:    [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[A]], <2 x i32> 
[[VSHRN_N]], <4 x i32> <i32 0, i32 1, i32 2, i32 3>
-// CHECK-NEXT:    ret <4 x i32> [[SHUFFLE_I]]
-//
-int32x4_t test_vshrn_high_n_s64(int32x2_t a, int64x2_t b) {
-  return vshrn_high_n_s64(a, b, 19);
-}
-
-// CHECK-LABEL: define dso_local <16 x i8> @test_vshrn_high_n_u16(
-// CHECK-SAME: <8 x i8> noundef [[A:%.*]], <8 x i16> noundef [[B:%.*]]) 
#[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    [[TMP0:%.*]] = bitcast <8 x i16> [[B]] to <16 x i8>
-// CHECK-NEXT:    [[TMP1:%.*]] = bitcast <16 x i8> [[TMP0]] to <8 x i16>
-// CHECK-NEXT:    [[TMP2:%.*]] = lshr <8 x i16> [[TMP1]], splat (i16 3)
-// CHECK-NEXT:    [[VSHRN_N:%.*]] = trunc <8 x i16> [[TMP2]] to <8 x i8>
-// CHECK-NEXT:    [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[A]], <8 x i8> 
[[VSHRN_N]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 
7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
-// CHECK-NEXT:    ret <16 x i8> [[SHUFFLE_I]]
-//
-uint8x16_t test_vshrn_high_n_u16(uint8x8_t a, uint16x8_t b) {
-  return vshrn_high_n_u16(a, b, 3);
-}
-
-// CHECK-LABEL: define dso_local <8 x i16> @test_vshrn_high_n_u32(
-// CHECK-SAME: <4 x i16> noundef [[A:%.*]], <4 x i32> noundef [[B:%.*]]) 
#[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    [[TMP0:%.*]] = bitcast <4 x i32> [[B]] to <16 x i8>
-// CHECK-NEXT:    [[TMP1:%.*]] = bitcast <16 x i8> [[TMP0]] to <4 x i32>
-// CHECK-NEXT:    [[TMP2:%.*]] = lshr <4 x i32> [[TMP1]], splat (i32 9)
-// CHECK-NEXT:    [[VSHRN_N:%.*]] = trunc <4 x i32> [[TMP2]] to <4 x i16>
-// CHECK-NEXT:    [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[A]], <4 x i16> 
[[VSHRN_N]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
-// CHECK-NEXT:    ret <8 x i16> [[SHUFFLE_I]]
-//
-uint16x8_t test_vshrn_high_n_u32(uint16x4_t a, uint32x4_t b) {
-  return vshrn_high_n_u32(a, b, 9);
-}
-
-// CHECK-LABEL: define dso_local <4 x i32> @test_vshrn_high_n_u64(
-// CHECK-SAME: <2 x i32> noundef [[A:%.*]], <2 x i64> noundef [[B:%.*]]) 
#[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    [[TMP0:%.*]] = bitcast <2 x i64> [[B]] to <16 x i8>
-// CHECK-NEXT:    [[TMP1:%.*]] = bitcast <16 x i8> [[TMP0]] to <2 x i64>
-// CHECK-NEXT:    [[TMP2:%.*]] = lshr <2 x i64> [[TMP1]], splat (i64 19)
-// CHECK-NEXT:    [[VSHRN_N:%.*]] = trunc <2 x i64> [[TMP2]] to <2 x i32>
-// CHECK-NEXT:    [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[A]], <2 x i32> 
[[VSHRN_N]], <4 x i32> <i32 0, i32 1, i32 2, i32 3>
-// CHECK-NEXT:    ret <4 x i32> [[SHUFFLE_I]]
-//
-uint32x4_t test_vshrn_high_n_u64(uint32x2_t a, uint64x2_t b) {
-  return vshrn_high_n_u64(a, b, 19);
-}
-
 // CHECK-LABEL: define dso_local <8 x i8> @test_vrshrn_n_s16(
 // CHECK-SAME: <8 x i16> noundef [[A:%.*]]) #[[ATTR0]] {
 // CHECK-NEXT:  [[ENTRY:.*:]]
diff --git a/clang/test/CodeGen/AArch64/neon/intrinsics.c 
b/clang/test/CodeGen/AArch64/neon/intrinsics.c
index 7175c616cf484f..3237c40dccb9ff 100644
--- a/clang/test/CodeGen/AArch64/neon/intrinsics.c
+++ b/clang/test/CodeGen/AArch64/neon/intrinsics.c
@@ -9110,6 +9110,179 @@ uint64x2_t test_vshll_high_n_u32(uint32x4_t a) {
   return vshll_high_n_u32(a, 19);
 }
 
+//===------------------------------------------------------===//
+// 2.1.3.2.5. Vector shift right and narrow
+// 
https://arm-software.github.io/acle/neon_intrinsics/advsimd.html#vector-shift-right-and-narrow
+//===------------------------------------------------------===//
+
+// ALL-LABEL: @test_vshrn_n_s16(
+int8x8_t test_vshrn_n_s16(int16x8_t a) {
+// CIR: [[TGT:%.*]] = cir.cast bitcast %{{.*}} : !cir.vector<16 x !s8i> -> 
!cir.vector<8 x !s16i>
+// CIR: [[SHIFT:%.*]] = cir.shift(right, [[TGT]] : !cir.vector<8 x !s16i>, 
%{{.*}} : !cir.vector<8 x !s16i>) -> !cir.vector<8 x !s16i>
+// CIR: cir.cast integral [[SHIFT]] : !cir.vector<8 x !s16i> -> !cir.vector<8 
x !s8i>
+
+// LLVM-SAME: <8 x i16> {{.*}} [[A:%.*]])
+// LLVM: [[SHIFT:%.*]] = ashr <8 x i16> {{%.*}}, splat (i16 3)
+// LLVM: [[NARROW:%.*]] = trunc <8 x i16> [[SHIFT]] to <8 x i8>
+// LLVM: ret <8 x i8> [[NARROW]]
+  return vshrn_n_s16(a, 3);
+}
+
+// ALL-LABEL: @test_vshrn_n_s32(
+int16x4_t test_vshrn_n_s32(int32x4_t a) {
+// CIR: [[TGT:%.*]] = cir.cast bitcast %{{.*}} : !cir.vector<16 x !s8i> -> 
!cir.vector<4 x !s32i>
+// CIR: [[SHIFT:%.*]] = cir.shift(right, [[TGT]] : !cir.vector<4 x !s32i>, 
%{{.*}} : !cir.vector<4 x !s32i>) -> !cir.vector<4 x !s32i>
+// CIR: cir.cast integral [[SHIFT]] : !cir.vector<4 x !s32i> -> !cir.vector<4 
x !s16i>
+
+// LLVM-SAME: <4 x i32> {{.*}} [[A:%.*]])
+// LLVM: [[SHIFT:%.*]] = ashr <4 x i32> {{%.*}}, splat (i32 9)
+// LLVM: [[NARROW:%.*]] = trunc <4 x i32> [[SHIFT]] to <4 x i16>
+// LLVM: ret <4 x i16> [[NARROW]]
+  return vshrn_n_s32(a, 9);
+}
+
+// ALL-LABEL: @test_vshrn_n_s64(
+int32x2_t test_vshrn_n_s64(int64x2_t a) {
+// CIR: [[TGT:%.*]] = cir.cast bitcast %{{.*}} : !cir.vector<16 x !s8i> -> 
!cir.vector<2 x !s64i>
+// CIR: [[SHIFT:%.*]] = cir.shift(right, [[TGT]] : !cir.vector<2 x !s64i>, 
%{{.*}} : !cir.vector<2 x !s64i>) -> !cir.vector<2 x !s64i>
+// CIR: cir.cast integral [[SHIFT]] : !cir.vector<2 x !s64i> -> !cir.vector<2 
x !s32i>
+
+// LLVM-SAME: <2 x i64> {{.*}} [[A:%.*]])
+// LLVM: [[SHIFT:%.*]] = ashr <2 x i64> {{%.*}}, splat (i64 19)
+// LLVM: [[NARROW:%.*]] = trunc <2 x i64> [[SHIFT]] to <2 x i32>
+// LLVM: ret <2 x i32> [[NARROW]]
+  return vshrn_n_s64(a, 19);
+}
+
+// ALL-LABEL: @test_vshrn_n_u16(
+uint8x8_t test_vshrn_n_u16(uint16x8_t a) {
+// CIR: [[TGT:%.*]] = cir.cast bitcast %{{.*}} : !cir.vector<16 x !s8i> -> 
!cir.vector<8 x !u16i>
+// CIR: [[SHIFT:%.*]] = cir.shift(right, [[TGT]] : !cir.vector<8 x !u16i>, 
%{{.*}} : !cir.vector<8 x !u16i>) -> !cir.vector<8 x !u16i>
+// CIR: cir.cast integral [[SHIFT]] : !cir.vector<8 x !u16i> -> !cir.vector<8 
x !u8i>
+
+// LLVM-SAME: <8 x i16> {{.*}} [[A:%.*]])
+// LLVM: [[SHIFT:%.*]] = lshr <8 x i16> {{%.*}}, splat (i16 3)
+// LLVM: [[NARROW:%.*]] = trunc <8 x i16> [[SHIFT]] to <8 x i8>
+// LLVM: ret <8 x i8> [[NARROW]]
+  return vshrn_n_u16(a, 3);
+}
+
+// ALL-LABEL: @test_vshrn_n_u32(
+uint16x4_t test_vshrn_n_u32(uint32x4_t a) {
+// CIR: [[TGT:%.*]] = cir.cast bitcast %{{.*}} : !cir.vector<16 x !s8i> -> 
!cir.vector<4 x !u32i>
+// CIR: [[SHIFT:%.*]] = cir.shift(right, [[TGT]] : !cir.vector<4 x !u32i>, 
%{{.*}} : !cir.vector<4 x !u32i>) -> !cir.vector<4 x !u32i>
+// CIR: cir.cast integral [[SHIFT]] : !cir.vector<4 x !u32i> -> !cir.vector<4 
x !u16i>
+
+// LLVM-SAME: <4 x i32> {{.*}} [[A:%.*]])
+// LLVM: [[SHIFT:%.*]] = lshr <4 x i32> {{%.*}}, splat (i32 9)
+// LLVM: [[NARROW:%.*]] = trunc <4 x i32> [[SHIFT]] to <4 x i16>
+// LLVM: ret <4 x i16> [[NARROW]]
+  return vshrn_n_u32(a, 9);
+}
+
+// ALL-LABEL: @test_vshrn_n_u64(
+uint32x2_t test_vshrn_n_u64(uint64x2_t a) {
+// CIR: [[TGT:%.*]] = cir.cast bitcast %{{.*}} : !cir.vector<16 x !s8i> -> 
!cir.vector<2 x !u64i>
+// CIR: [[SHIFT:%.*]] = cir.shift(right, [[TGT]] : !cir.vector<2 x !u64i>, 
%{{.*}} : !cir.vector<2 x !u64i>) -> !cir.vector<2 x !u64i>
+// CIR: cir.cast integral [[SHIFT]] : !cir.vector<2 x !u64i> -> !cir.vector<2 
x !u32i>
+
+// LLVM-SAME: <2 x i64> {{.*}} [[A:%.*]])
+// LLVM: [[SHIFT:%.*]] = lshr <2 x i64> {{%.*}}, splat (i64 19)
+// LLVM: [[NARROW:%.*]] = trunc <2 x i64> [[SHIFT]] to <2 x i32>
+// LLVM: ret <2 x i32> [[NARROW]]
+  return vshrn_n_u64(a, 19);
+}
+
+// ALL-LABEL: @test_vshrn_high_n_s16(
+int8x16_t test_vshrn_high_n_s16(int8x8_t a, int16x8_t b) {
+// CIR: [[TGT:%.*]] = cir.cast bitcast %{{.*}} : !cir.vector<16 x !s8i> -> 
!cir.vector<8 x !s16i>
+// CIR: [[SHIFT:%.*]] = cir.shift(right, [[TGT]] : !cir.vector<8 x !s16i>, 
%{{.*}} : !cir.vector<8 x !s16i>) -> !cir.vector<8 x !s16i>
+// CIR: cir.cast integral [[SHIFT]] : !cir.vector<8 x !s16i> -> !cir.vector<8 
x !s8i>
+// CIR: cir.call @vcombine_s8
+
+// LLVM-SAME: <8 x i8> {{.*}} [[LOW:%.*]], <8 x i16> {{.*}} [[A:%.*]])
+// LLVM: [[SHIFT:%.*]] = ashr <8 x i16> {{%.*}}, splat (i16 3)
+// LLVM: [[NARROW:%.*]] = trunc <8 x i16> [[SHIFT]] to <8 x i8>
+// LLVM: [[COMBINED:%.*]] = shufflevector <8 x i8> [[LOW]], <8 x i8> 
[[NARROW]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, 
i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+// LLVM: ret <16 x i8> [[COMBINED]]
+  return vshrn_high_n_s16(a, b, 3);
+}
+
+// ALL-LABEL: @test_vshrn_high_n_s32(
+int16x8_t test_vshrn_high_n_s32(int16x4_t a, int32x4_t b) {
+// CIR: [[TGT:%.*]] = cir.cast bitcast %{{.*}} : !cir.vector<16 x !s8i> -> 
!cir.vector<4 x !s32i>
+// CIR: [[SHIFT:%.*]] = cir.shift(right, [[TGT]] : !cir.vector<4 x !s32i>, 
%{{.*}} : !cir.vector<4 x !s32i>) -> !cir.vector<4 x !s32i>
+// CIR: cir.cast integral [[SHIFT]] : !cir.vector<4 x !s32i> -> !cir.vector<4 
x !s16i>
+// CIR: cir.call @vcombine_s16
+
+// LLVM-SAME: <4 x i16> {{.*}} [[LOW:%.*]], <4 x i32> {{.*}} [[A:%.*]])
+// LLVM: [[SHIFT:%.*]] = ashr <4 x i32> {{%.*}}, splat (i32 9)
+// LLVM: [[NARROW:%.*]] = trunc <4 x i32> [[SHIFT]] to <4 x i16>
+// LLVM: [[COMBINED:%.*]] = shufflevector <4 x i16> [[LOW]], <4 x i16> 
[[NARROW]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+// LLVM: ret <8 x i16> [[COMBINED]]
+  return vshrn_high_n_s32(a, b, 9);
+}
+
+// ALL-LABEL: @test_vshrn_high_n_s64(
+int32x4_t test_vshrn_high_n_s64(int32x2_t a, int64x2_t b) {
+// CIR: [[TGT:%.*]] = cir.cast bitcast %{{.*}} : !cir.vector<16 x !s8i> -> 
!cir.vector<2 x !s64i>
+// CIR: [[SHIFT:%.*]] = cir.shift(right, [[TGT]] : !cir.vector<2 x !s64i>, 
%{{.*}} : !cir.vector<2 x !s64i>) -> !cir.vector<2 x !s64i>
+// CIR: cir.cast integral [[SHIFT]] : !cir.vector<2 x !s64i> -> !cir.vector<2 
x !s32i>
+// CIR: cir.call @vcombine_s32
+
+// LLVM-SAME: <2 x i32> {{.*}} [[LOW:%.*]], <2 x i64> {{.*}} [[A:%.*]])
+// LLVM: [[SHIFT:%.*]] = ashr <2 x i64> {{%.*}}, splat (i64 19)
+// LLVM: [[NARROW:%.*]] = trunc <2 x i64> [[SHIFT]] to <2 x i32>
+// LLVM: [[COMBINED:%.*]] = shufflevector <2 x i32> [[LOW]], <2 x i32> 
[[NARROW]], <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+// LLVM: ret <4 x i32> [[COMBINED]]
+  return vshrn_high_n_s64(a, b, 19);
+}
+
+// ALL-LABEL: @test_vshrn_high_n_u16(
+uint8x16_t test_vshrn_high_n_u16(uint8x8_t a, uint16x8_t b) {
+// CIR: [[TGT:%.*]] = cir.cast bitcast %{{.*}} : !cir.vector<16 x !s8i> -> 
!cir.vector<8 x !u16i>
+// CIR: [[SHIFT:%.*]] = cir.shift(right, [[TGT]] : !cir.vector<8 x !u16i>, 
%{{.*}} : !cir.vector<8 x !u16i>) -> !cir.vector<8 x !u16i>
+// CIR: cir.cast integral [[SHIFT]] : !cir.vector<8 x !u16i> -> !cir.vector<8 
x !u8i>
+// CIR: cir.call @vcombine_u8
+
+// LLVM-SAME: <8 x i8> {{.*}} [[LOW:%.*]], <8 x i16> {{.*}} [[A:%.*]])
+// LLVM: [[SHIFT:%.*]] = lshr <8 x i16> {{%.*}}, splat (i16 3)
+// LLVM: [[NARROW:%.*]] = trunc <8 x i16> [[SHIFT]] to <8 x i8>
+// LLVM: [[COMBINED:%.*]] = shufflevector <8 x i8> [[LOW]], <8 x i8> 
[[NARROW]], <16 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, 
i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14, i32 15>
+// LLVM: ret <16 x i8> [[COMBINED]]
+  return vshrn_high_n_u16(a, b, 3);
+}
+
+// ALL-LABEL: @test_vshrn_high_n_u32(
+uint16x8_t test_vshrn_high_n_u32(uint16x4_t a, uint32x4_t b) {
+// CIR: [[TGT:%.*]] = cir.cast bitcast %{{.*}} : !cir.vector<16 x !s8i> -> 
!cir.vector<4 x !u32i>
+// CIR: [[SHIFT:%.*]] = cir.shift(right, [[TGT]] : !cir.vector<4 x !u32i>, 
%{{.*}} : !cir.vector<4 x !u32i>) -> !cir.vector<4 x !u32i>
+// CIR: cir.cast integral [[SHIFT]] : !cir.vector<4 x !u32i> -> !cir.vector<4 
x !u16i>
+// CIR: cir.call @vcombine_u16
+
+// LLVM-SAME: <4 x i16> {{.*}} [[LOW:%.*]], <4 x i32> {{.*}} [[A:%.*]])
+// LLVM: [[SHIFT:%.*]] = lshr <4 x i32> {{%.*}}, splat (i32 9)
+// LLVM: [[NARROW:%.*]] = trunc <4 x i32> [[SHIFT]] to <4 x i16>
+// LLVM: [[COMBINED:%.*]] = shufflevector <4 x i16> [[LOW]], <4 x i16> 
[[NARROW]], <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+// LLVM: ret <8 x i16> [[COMBINED]]
+  return vshrn_high_n_u32(a, b, 9);
+}
+
+// ALL-LABEL: @test_vshrn_high_n_u64(
+uint32x4_t test_vshrn_high_n_u64(uint32x2_t a, uint64x2_t b) {
+// CIR: [[TGT:%.*]] = cir.cast bitcast %{{.*}} : !cir.vector<16 x !s8i> -> 
!cir.vector<2 x !u64i>
+// CIR: [[SHIFT:%.*]] = cir.shift(right, [[TGT]] : !cir.vector<2 x !u64i>, 
%{{.*}} : !cir.vector<2 x !u64i>) -> !cir.vector<2 x !u64i>
+// CIR: cir.cast integral [[SHIFT]] : !cir.vector<2 x !u64i> -> !cir.vector<2 
x !u32i>
+// CIR: cir.call @vcombine_u32
+
+// LLVM-SAME: <2 x i32> {{.*}} [[LOW:%.*]], <2 x i64> {{.*}} [[A:%.*]])
+// LLVM: [[SHIFT:%.*]] = lshr <2 x i64> {{%.*}}, splat (i64 19)
+// LLVM: [[NARROW:%.*]] = trunc <2 x i64> [[SHIFT]] to <2 x i32>
+// LLVM: [[COMBINED:%.*]] = shufflevector <2 x i32> [[LOW]], <2 x i32> 
[[NARROW]], <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+// LLVM: ret <4 x i32> [[COMBINED]]
+  return vshrn_high_n_u64(a, b, 19);
+}
+
 //===------------------------------------------------------===//
 // 2.1.3.2.6. Vector saturating shift right and narrow
 // 
https://arm-software.github.io/acle/neon_intrinsics/advsimd.html#vector-saturating-shift-right-and-narrow

_______________________________________________
cfe-commits mailing list
[email protected]
https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits

Reply via email to