https://github.com/topperc updated https://github.com/llvm/llvm-project/pull/224739
>From e7d9098204fd16c1399152915147d8e3b08bad4c Mon Sep 17 00:00:00 2001 From: Craig Topper <[email protected]> Date: Fri, 18 Sep 2026 13:22:10 -0700 Subject: [PATCH 1/4] [RISCV] Add a command line option to disable register overlap for vector index load. If a trap occurs on an element other than 0, some SiFive cores will write a garbage value to that element and all active elements after it in the destination register before running the trap handler. This is normally allowed by the spec if restarting the instruction would overwrite the elements with the correct results. For a vector indexed load with overlapping source and destination, writing the later elements with garbage corrupts some of the source indices. This prevents the instruction from restarting correctly once the fault has been handled. This option allows users to opt out of overlapping registers on these instructions. This is an alternative to #220079. I have not enabled this by default for -mcpu yet. --- clang/include/clang/Options/Options.td | 4 ++++ clang/lib/Driver/ToolChains/Arch/RISCV.cpp | 8 ++++++++ clang/test/Driver/riscv-features.c | 5 +++++ llvm/lib/Target/RISCV/RISCVFeatures.td | 4 ++++ llvm/lib/Target/RISCV/RISCVISelLowering.cpp | 14 ++++++++++++++ llvm/lib/Target/RISCV/RISCVInstrInfoVPseudos.td | 2 ++ llvm/test/CodeGen/RISCV/features-info.ll | 1 + 7 files changed, 38 insertions(+) diff --git a/clang/include/clang/Options/Options.td b/clang/include/clang/Options/Options.td index c4c49df73c15a..4a6e77e204412 100644 --- a/clang/include/clang/Options/Options.td +++ b/clang/include/clang/Options/Options.td @@ -5855,6 +5855,10 @@ def mzilsd_word_align : Flag<["-"], "mzilsd-word-align">, Group<m_Group>, HelpText<"Allow Zilsd/Zclsd memory accesses to be 4-byte aligned (RISC-V only)">; def mzilsd_strict_align : Flag<["-"], "mzilsd-strict-align">, Group<m_Group>, HelpText<"Force all Zilsd/Zclsd memory accesses to be 8-byte aligned (RISC-V only)">; +def mvector_index_load_overlap : Flag<["-"], "mvector-index-load-overlap">, Group<m_Group>, + HelpText<"Allow vector index load to have overlapping source and destination register groups (RISC-V only)">; +def mno_vector_index_load_overlap : Flag<["-"], "mno-vector-index-load-overlap">, Group<m_Group>, + HelpText<"Force vector index load to have non-overlapping source and destination register groups (RISC-V only)">; def mno_thumb : Flag<["-"], "mno-thumb">, Group<m_arm_Features_Group>; def mrestrict_it: Flag<["-"], "mrestrict-it">, Group<m_arm_Features_Group>, HelpText<"Disallow generation of complex IT blocks. It is off by default.">; diff --git a/clang/lib/Driver/ToolChains/Arch/RISCV.cpp b/clang/lib/Driver/ToolChains/Arch/RISCV.cpp index 8e650ddf92dfc..2a55754c1aac3 100644 --- a/clang/lib/Driver/ToolChains/Arch/RISCV.cpp +++ b/clang/lib/Driver/ToolChains/Arch/RISCV.cpp @@ -178,6 +178,14 @@ void riscv::getRISCVTargetFeatures(const Driver &D, const llvm::Triple &Triple, Features.push_back("+unaligned-vector-mem"); } + if (const Arg *A = + Args.getLastArg(options::OPT_mvector_index_load_overlap, + options::OPT_mno_vector_index_load_overlap)) { + if (A->getOption().matches(options::OPT_mno_vector_index_load_overlap)) + Features.push_back("+no-vector-index-load-overlap"); + else + Features.push_back("-no-vector-index-load-overlap"); + } if (Triple.isRISCV32()) { // Handle `-mzilsd-word-align` and `-mzilsd-strict-align` on rv32. These // interact with the scalar alignment options - if unaligned scalar memory diff --git a/clang/test/Driver/riscv-features.c b/clang/test/Driver/riscv-features.c index 97736ff81c799..287077c7cd18d 100644 --- a/clang/test/Driver/riscv-features.c +++ b/clang/test/Driver/riscv-features.c @@ -48,6 +48,11 @@ // FAST-VECTOR-UNALIGNED-ACCESS: "-target-feature" "+unaligned-vector-mem" // NO-FAST-VECTOR-UNALIGNED-ACCESS: "-target-feature" "-unaligned-vector-mem" +// RUN: %clang --target=riscv32-unknown-elf -### %s -mvector-index-load-overlap 2>&1 | FileCheck %s -check-prefix=VECTOR-INDEX-LOAD-OVERLAP +// RUN: %clang --target=riscv32-unknown-elf -### %s -mno-vector-index-load-overlap 2>&1 | FileCheck %s -check-prefix=NO-VECTOR-INDEX-LOAD-OVERLAP +// VECTOR-INDEX-LOAD-OVERLAP: "-target-feature" "-no-vector-index-load-overlap" +// NO-VECTOR-INDEX-LOAD-OVERLAP: "-target-feature" "+no-vector-index-load-overlap" + // RUN: %clang --target=riscv32-unknown-elf -### %s 2>&1 | FileCheck %s -check-prefix=NOUWTABLE // RUN: %clang --target=riscv32-unknown-elf -fasynchronous-unwind-tables -### %s 2>&1 | FileCheck %s -check-prefix=UWTABLE // RUN: %clang --target=riscv64-unknown-elf -### %s 2>&1 | FileCheck %s -check-prefix=NOUWTABLE diff --git a/llvm/lib/Target/RISCV/RISCVFeatures.td b/llvm/lib/Target/RISCV/RISCVFeatures.td index ca68611339ee1..4376721086bf8 100644 --- a/llvm/lib/Target/RISCV/RISCVFeatures.td +++ b/llvm/lib/Target/RISCV/RISCVFeatures.td @@ -2057,6 +2057,10 @@ def FeatureTaggedGlobals : SubtargetFeature<"tagged-globals", "true", "Use an instruction sequence for taking the address of a global " "that allows a memory tag in the upper address bits">; +def FeatureNoVectorIndexLoadOverlap : SubtargetFeature<"no-vector-index-load-overlap", + "AllowVectorIndexLoadOverlap", "false", + "Disable overlapping source and destination for vector index loads">; + //===----------------------------------------------------------------------===// // Tuning features //===----------------------------------------------------------------------===// diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp index e749279168ab9..7d25baea79fd5 100644 --- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp +++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp @@ -27003,6 +27003,20 @@ RISCVTargetLowering::EmitInstrWithCustomInserter(MachineInstr &MI, void RISCVTargetLowering::AdjustInstrPostInstrSelection(MachineInstr &MI, SDNode *Node) const { + if (!Subtarget.allowVectorIndexLoadOverlap()) { + const RISCVVPseudosTable::PseudoInfo *RVV = + RISCVVPseudosTable::getPseudoInfo(MI.getOpcode()); + if (RVV && (RVV->BaseInstr == RISCV::VLUXEI8_V || + RVV->BaseInstr == RISCV::VLUXEI16_V || + RVV->BaseInstr == RISCV::VLUXEI32_V || + RVV->BaseInstr == RISCV::VLUXEI64_V || + RVV->BaseInstr == RISCV::VLOXEI8_V || + RVV->BaseInstr == RISCV::VLOXEI16_V || + RVV->BaseInstr == RISCV::VLOXEI32_V || + RVV->BaseInstr == RISCV::VLOXEI64_V)) + MI.getOperand(0).setIsEarlyClobber(true); + } + // If instruction defines FRM operand, conservatively set it as non-dead to // express data dependency with FRM users and prevent incorrect instruction // reordering. diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoVPseudos.td b/llvm/lib/Target/RISCV/RISCVInstrInfoVPseudos.td index c6241f582de67..e2fc58234e208 100644 --- a/llvm/lib/Target/RISCV/RISCVInstrInfoVPseudos.td +++ b/llvm/lib/Target/RISCV/RISCVInstrInfoVPseudos.td @@ -936,6 +936,7 @@ class VPseudoILoadNoMask<VReg RetClass, let HasVecPolicyOp = 1; let Constraints = !if(!eq(EarlyClobber, 1), "@earlyclobber $rd, $rd = $passthru", "$rd = $passthru"); let TargetOverlapConstraintType = TargetConstraintType; + let hasPostISelHook = 1; } class VPseudoILoadMask<VReg RetClass, @@ -960,6 +961,7 @@ class VPseudoILoadMask<VReg RetClass, let HasVecPolicyOp = 1; let UsesMaskPolicy = 1; let IncludeInInversePseudoTable = 0; + let hasPostISelHook = 1; } class VPseudoUSStoreNoMask<VReg StClass, diff --git a/llvm/test/CodeGen/RISCV/features-info.ll b/llvm/test/CodeGen/RISCV/features-info.ll index 6d4a437595be8..7b33fc78cc605 100644 --- a/llvm/test/CodeGen/RISCV/features-info.ll +++ b/llvm/test/CodeGen/RISCV/features-info.ll @@ -102,6 +102,7 @@ ; CHECK-NEXT: no-default-unroll - Disable default unroll preference. ; CHECK-NEXT: no-sink-splat-operands - Disable sink splat operands to enable .vx, .vf,.wx, and .wf instructions. ; CHECK-NEXT: no-trailing-seq-cst-fence - Disable trailing fence for seq-cst store. +; CHECK-NEXT: no-vector-index-load-overlap - Disable overlapping source and destination for vector index loads. ; CHECK-NEXT: optimized-nf2-segment-load-store - vlseg2eN.v and vsseg2eN.v are implemented as a wide memory op and shuffle. ; CHECK-NEXT: optimized-nf3-segment-load-store - vlseg3eN.v and vsseg3eN.v are implemented as a wide memory op and shuffle. ; CHECK-NEXT: optimized-nf4-segment-load-store - vlseg4eN.v and vsseg4eN.v are implemented as a wide memory op and shuffle. >From a534d1d20fdb2da671f0ff4e7b83a1413bd19efd Mon Sep 17 00:00:00 2001 From: Craig Topper <[email protected]> Date: Fri, 18 Sep 2026 15:10:51 -0700 Subject: [PATCH 2/4] fixup! Add tests --- .../RISCV/rvv/vluxei-vloxei-overlap.ll | 58 +++++++++++++++++++ 1 file changed, 58 insertions(+) create mode 100644 llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll diff --git a/llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll b/llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll new file mode 100644 index 0000000000000..f8d945d1d5eaa --- /dev/null +++ b/llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll @@ -0,0 +1,58 @@ +; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py +; RUN: sed 's/iXLen/i32/g' %s | llc -mtriple=riscv32 -mattr=+v,+zvfhmin,+zvfbfmin \ +; RUN: -verify-machineinstrs -target-abi=ilp32d | FileCheck %s --check-prefixes=OVERLAP +; RUN: sed 's/iXLen/i64/g' %s | llc -mtriple=riscv64 -mattr=+v,+zvfhmin,+zvfbfmin \ +; RUN: -verify-machineinstrs -target-abi=lp64d | FileCheck %s --check-prefixes=OVERLAP +; RUN: sed 's/iXLen/i32/g' %s | llc -mtriple=riscv32 -mattr=+v,+zvfhmin,+zvfbfmin,+no-vector-index-load-overlap \ +; RUN: -verify-machineinstrs -target-abi=ilp32d | FileCheck %s --check-prefixes=NOOVERLAP +; RUN: sed 's/iXLen/i64/g' %s | llc -mtriple=riscv64 -mattr=+v,+zvfhmin,+zvfbfmin,+no-vector-index-load-overlap \ +; RUN: -verify-machineinstrs -target-abi=lp64d | FileCheck %s --check-prefixes=NOOVERLAP + +define <vscale x 1 x i8> @intrinsic_vluxei_v_nxv1i8_nxv1i8_nxv1i32(ptr %0, <vscale x 1 x i32> %1, iXLen %2) nounwind { +; OVERLAP-LABEL: intrinsic_vluxei_v_nxv1i8_nxv1i8_nxv1i32: +; OVERLAP: # %bb.0: # %entry +; OVERLAP-NEXT: vsetvli zero, a1, e8, mf8, ta, ma +; OVERLAP-NEXT: vluxei32.v v8, (a0), v8 +; OVERLAP-NEXT: ret +; +; NOOVERLAP-LABEL: intrinsic_vluxei_v_nxv1i8_nxv1i8_nxv1i32: +; NOOVERLAP: # %bb.0: # %entry +; NOOVERLAP-NEXT: vsetvli zero, a1, e8, mf8, ta, ma +; NOOVERLAP-NEXT: vluxei32.v v9, (a0), v8 +; NOOVERLAP-NEXT: vmv1r.v v8, v9 +; NOOVERLAP-NEXT: ret +entry: + %a = call <vscale x 1 x i8> @llvm.riscv.vluxei.nxv1i8.nxv1i32( + <vscale x 1 x i8> poison, + ptr %0, + <vscale x 1 x i32> %1, + iXLen %2) + + ret <vscale x 1 x i8> %a +} + +define <vscale x 1 x i8> @intrinsic_vloxei_v_nxv1i8_nxv1i8_nxv1i32(ptr %0, <vscale x 1 x i32> %1, iXLen %2) nounwind { +; OVERLAP-LABEL: intrinsic_vloxei_v_nxv1i8_nxv1i8_nxv1i32: +; OVERLAP: # %bb.0: # %entry +; OVERLAP-NEXT: vsetvli zero, a1, e8, mf8, ta, ma +; OVERLAP-NEXT: vloxei32.v v8, (a0), v8 +; OVERLAP-NEXT: ret +; +; NOOVERLAP-LABEL: intrinsic_vloxei_v_nxv1i8_nxv1i8_nxv1i32: +; NOOVERLAP: # %bb.0: # %entry +; NOOVERLAP-NEXT: vsetvli zero, a1, e8, mf8, ta, ma +; NOOVERLAP-NEXT: vloxei32.v v9, (a0), v8 +; NOOVERLAP-NEXT: vmv1r.v v8, v9 +; NOOVERLAP-NEXT: ret +entry: + %a = call <vscale x 1 x i8> @llvm.riscv.vloxei.nxv1i8.nxv1i32( + <vscale x 1 x i8> poison, + ptr %0, + <vscale x 1 x i32> %1, + iXLen %2) + + ret <vscale x 1 x i8> %a +} + +declare <vscale x 1 x i8> @llvm.riscv.vluxei.nxv1i8.nxv1i32(<vscale x 1 x i8>, ptr, <vscale x 1 x i32>, iXLen) +declare <vscale x 1 x i8> @llvm.riscv.vloxei.nxv1i8.nxv1i32(<vscale x 1 x i8>, ptr, <vscale x 1 x i32>, iXLen) >From 59090ba91bc5989fa82069b677a460b937742a91 Mon Sep 17 00:00:00 2001 From: Craig Topper <[email protected]> Date: Mon, 21 Sep 2026 12:32:24 -0700 Subject: [PATCH 3/4] fixup! Drop intrinsic declarations --- llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll | 3 --- 1 file changed, 3 deletions(-) diff --git a/llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll b/llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll index f8d945d1d5eaa..dfe500f050bc6 100644 --- a/llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll +++ b/llvm/test/CodeGen/RISCV/rvv/vluxei-vloxei-overlap.ll @@ -53,6 +53,3 @@ entry: ret <vscale x 1 x i8> %a } - -declare <vscale x 1 x i8> @llvm.riscv.vluxei.nxv1i8.nxv1i32(<vscale x 1 x i8>, ptr, <vscale x 1 x i32>, iXLen) -declare <vscale x 1 x i8> @llvm.riscv.vloxei.nxv1i8.nxv1i32(<vscale x 1 x i8>, ptr, <vscale x 1 x i32>, iXLen) >From 3985e40bcf8fa5d572af197427757ece8a7d2658 Mon Sep 17 00:00:00 2001 From: Craig Topper <[email protected]> Date: Mon, 21 Sep 2026 15:57:38 -0700 Subject: [PATCH 4/4] fixup! Use BoolOptionWithoutMarshalling. --- clang/include/clang/Options/Options.td | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/clang/include/clang/Options/Options.td b/clang/include/clang/Options/Options.td index 4a6e77e204412..6e8c5f4226bbf 100644 --- a/clang/include/clang/Options/Options.td +++ b/clang/include/clang/Options/Options.td @@ -5855,10 +5855,11 @@ def mzilsd_word_align : Flag<["-"], "mzilsd-word-align">, Group<m_Group>, HelpText<"Allow Zilsd/Zclsd memory accesses to be 4-byte aligned (RISC-V only)">; def mzilsd_strict_align : Flag<["-"], "mzilsd-strict-align">, Group<m_Group>, HelpText<"Force all Zilsd/Zclsd memory accesses to be 8-byte aligned (RISC-V only)">; -def mvector_index_load_overlap : Flag<["-"], "mvector-index-load-overlap">, Group<m_Group>, - HelpText<"Allow vector index load to have overlapping source and destination register groups (RISC-V only)">; -def mno_vector_index_load_overlap : Flag<["-"], "mno-vector-index-load-overlap">, Group<m_Group>, - HelpText<"Force vector index load to have non-overlapping source and destination register groups (RISC-V only)">; +defm vector_index_load_overlap : BoolOptionWithoutMarshalling<"m", "vector-index-load-overlap", + PosFlag<SetTrue, [], [], "Allow vector index load to have ">, + NegFlag<SetFalse, [], [], "Force vector index load to have non-">, + BothFlags<[], [], "overlapping source and destination register groups (RISC-V only)">>, + Group<m_Group>; def mno_thumb : Flag<["-"], "mno-thumb">, Group<m_arm_Features_Group>; def mrestrict_it: Flag<["-"], "mrestrict-it">, Group<m_arm_Features_Group>, HelpText<"Disallow generation of complex IT blocks. It is off by default.">; _______________________________________________ cfe-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits
