Author: Hans Wennborg Date: 2026-09-02T14:59:18+02:00 New Revision: 58b734b38ae1b151c2f012006227ef76e763b016
URL: https://github.com/llvm/llvm-project/commit/58b734b38ae1b151c2f012006227ef76e763b016 DIFF: https://github.com/llvm/llvm-project/commit/58b734b38ae1b151c2f012006227ef76e763b016.diff LOG: Revert "[SLP]Vectorize unique scalars of splat gather nodes as separate subtrees" This reverts commit aa82f051348076b2f6fdcfb3ae616b34700b17af. Added: Modified: llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp llvm/test/Transforms/SLPVectorizer/AArch64/loadi8.ll llvm/test/Transforms/SLPVectorizer/AArch64/matmul.ll llvm/test/Transforms/SLPVectorizer/AArch64/splat-loads.ll llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-extracts.ll llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-trim-combined.ll llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-trim.ll llvm/test/Transforms/SLPVectorizer/X86/broadcast.ll llvm/test/Transforms/SLPVectorizer/X86/lookahead.ll llvm/test/Transforms/SLPVectorizer/X86/reorder_phi.ll llvm/test/Transforms/SLPVectorizer/X86/splat-gather-operands.ll Removed: ################################################################################ diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp index a9681210c6eaf..91e42668ea508 100644 --- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp +++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp @@ -821,7 +821,6 @@ class slpvectorizer::BoUpSLP { LoadEntriesToVectorize.clear(); IsGraphTransformMode = false; GatheredLoadsEntriesFirst.reset(); - SplatGatheredScalarsRoots.clear(); CompressEntryToData.clear(); ExternalUses.clear(); ExternalUsesAsOriginalScalar.clear(); @@ -2776,12 +2775,6 @@ class slpvectorizer::BoUpSLP { SmallVector<SmallVector<std::pair<LoadInst *, int64_t>>>, 8> &GatheredLoads); - /// Run through the gather nodes that are splats of the same instruction and - /// try to vectorize the unique splatted values together as a separate - /// subtree. The splat gathers are then emitted as broadcasts of the - /// vectorized subtree instead of insertion sequences. - void tryToVectorizeSplatGatheredScalars(); - /// Helper for `findExternalStoreUsersReorderIndices()`. It iterates over the /// users of \p TE and collects the stores. It returns the map from the store /// pointers to the collected stores. @@ -3657,11 +3650,6 @@ class slpvectorizer::BoUpSLP { /// The index of the first gathered load entry in the VectorizeTree. std::optional<unsigned> GatheredLoadsEntriesFirst; - /// Root entries of the subtrees built for the splat gather nodes' unique - /// scalars. They have no users in the tree and must be emitted explicitly - /// before the root node. - SmallVector<TreeEntry *> SplatGatheredScalarsRoots; - /// Maps compress entries to their mask data for the final codegen. SmallDenseMap<const TreeEntry *, std::tuple<SmallVector<int>, VectorType *, unsigned, bool>> @@ -9046,8 +9034,6 @@ void BoUpSLP::buildTree(ArrayRef<Value *> Roots, if (!allSameType(Roots)) return; buildTreeRec(Roots, 0, EdgeInfo()); - // Build splat-gather subtrees here so the reordering passes cover them too. - tryToVectorizeSplatGatheredScalars(); } void BoUpSLP::buildTree(ArrayRef<Value *> Roots) { @@ -9057,8 +9043,6 @@ void BoUpSLP::buildTree(ArrayRef<Value *> Roots) { if (!allSameType(Roots)) return; buildTreeRec(Roots, 0, EdgeInfo()); - // Build splat-gather subtrees here so the reordering passes cover them too. - tryToVectorizeSplatGatheredScalars(); } /// Tries to find subvector of loads and builds new vector of only loads if can @@ -9744,12 +9728,10 @@ static std::pair<size_t, size_t> generateKeySubkey( if (isTriviallyVectorizable(ID)) { if (ID == Intrinsic::fmuladd) ID = Intrinsic::fma; - SubKey = hash_combine(hash_value(I->getOpcode()), hash_value(ID), - hash_value(I->getType())); + SubKey = hash_combine(hash_value(I->getOpcode()), hash_value(ID)); } else if (!VFDatabase(*Call).getMappings(*Call).empty()) { SubKey = hash_combine(hash_value(I->getOpcode()), - hash_value(Call->getCalledFunction()), - hash_value(I->getType())); + hash_value(Call->getCalledFunction())); } else { Key = hash_combine(hash_value(Call), Key); SubKey = hash_combine(hash_value(I->getOpcode()), hash_value(Call)); @@ -9767,8 +9749,7 @@ static std::pair<size_t, size_t> generateKeySubkey( // Do not try to vectorize instructions with potentially high cost. SubKey = hash_value(I); } else { - SubKey = - hash_combine(hash_value(I->getOpcode()), hash_value(I->getType())); + SubKey = hash_value(I->getOpcode()); } Key = hash_combine(hash_value(I->getParent()->getNumber()), Key); } @@ -12144,75 +12125,6 @@ class InstructionsCompatibilityAnalysis { }; } // namespace -void BoUpSLP::tryToVectorizeSplatGatheredScalars() { - auto LoadsSubkey = [](size_t /*Key*/, LoadInst *LI) { - return hash_value(getUnderlyingObject(LI->getPointerOperand())); - }; - SmallMapVector<std::pair<size_t, size_t>, SmallSetVector<Value *, 4>, 4> - Groups; - for (const std::unique_ptr<TreeEntry> &TE : VectorizableTree) { - // Only gathers with vectorized (non-gather) users can reuse the broadcast. - if (!TE->isGather() || !TE->UserTreeIndex || - TE->UserTreeIndex.UserTE->isGather() || !isSplat(TE->Scalars)) - continue; - auto *I = dyn_cast<Instruction>(TE->Scalars.front()); - // Skip shuffle-like instructions: their splat gathers are already emitted - // as cheap shuffles of the source vector. - if (!I || - isa<ExtractElementInst, InsertElementInst, ShuffleVectorInst>(I) || - I->getType()->isVoidTy() || isVectorized(I) || isDeleted(I) || - (UserIgnoreList && UserIgnoreList->contains(I))) - continue; - // Scheduling new memory bundles in a to-be-versioned tree records extra - // alias-check pairs and can push the region over the versioning limits. - if (isTryingRuntimeAliasChecks() && I->mayReadOrWriteMemory()) - continue; - Groups[generateKeySubkey(I, TLI, LoadsSubkey, /*AllowAlternate=*/true)] - .insert(I); - } - // Values left in singleton groups cannot form a bundle on their own; - // regroup them by the opcode-insensitive key so alternate/copyable - // bundles still form. - SmallMapVector<std::pair<size_t, Type *>, SmallSetVector<Value *, 4>, 4> - FallbackGroups; - for (auto &[Key, Group] : Groups) { - if (Group.size() >= 2) - continue; - FallbackGroups[std::make_pair(Key.first, Group.front()->getType())].insert( - Group.front()); - } - InstructionsCompatibilityAnalysis Analysis(*DT, *DL, *TTI, *TLI); - auto BuildSubtree = [&](const auto &GroupMap) { - for (const auto &[_, Group] : GroupMap) { - if (Group.size() < 2) - continue; - // Copyable-aware check so bundles with copyable lanes are not skipped. - if (!Analysis.buildInstructionsState(Group.getArrayRef(), *this)) - continue; - unsigned PrevSize = VectorizableTree.size(); - buildTreeRec(Group.getArrayRef(), 0, EdgeInfo()); - if (PrevSize == VectorizableTree.size()) - continue; - TreeEntry *NewRoot = VectorizableTree[PrevSize].get(); - if (NewRoot->isGather()) { - // Failed to vectorize the bundle: drop the added gather entry, it has - // no users and only adds cost. - for (Value *V : NewRoot->Scalars) { - auto It = ValueToGatherNodes.find(V); - if (It != ValueToGatherNodes.end()) - It->second.remove(NewRoot); - } - LoadEntriesToVectorize.remove(PrevSize); - VectorizableTree.pop_back(); - continue; - } - SplatGatheredScalarsRoots.push_back(NewRoot); - } - }; - BuildSubtree(Groups); - BuildSubtree(FallbackGroups); -} - BoUpSLP::ScalarsVectorizationLegality BoUpSLP::getScalarsVectorizationLegality(ArrayRef<Value *> VL, unsigned Depth, const EdgeInfo &UserTreeIdx) const { @@ -19482,11 +19394,8 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals, } } // Bail out if the cost threshold is negative and cost already below it. - // The splat subtrees may still force extracts of their scalars on top of - // the node cost and have to be trimmed, so do not bail out if there are - // any. if (SLPCostThreshold.getNumOccurrences() > 0 && SLPCostThreshold < 0 && - Cost < -SLPCostThreshold && SplatGatheredScalarsRoots.empty()) + Cost < -SLPCostThreshold) return Cost; // The narrow non-profitable tree in loop? Skip, may cause regressions. constexpr unsigned PartLimit = 2; @@ -19744,30 +19653,23 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals, } Worklist.pop(); } - if (!Changed) { - // The splat subtrees are not linked to the tree root, so their cost is - // not included in the root's subtree cost; add it explicitly. - InstructionCost TotalCost = std::get<1>(SubtreeCosts.front()); - for (const TreeEntry *TE : SplatGatheredScalarsRoots) - TotalCost += std::get<1>(SubtreeCosts[TE->Idx]); - return TotalCost; - } + if (!Changed) + return std::get<1>(SubtreeCosts.front()); - SmallPtrSet<TreeEntry *, 4> SubtreesToDelete; + SmallPtrSet<TreeEntry *, 4> GatheredLoadsToDelete; InstructionCost LoadsExtractsCost = 0; - using ValuesToInsertTy = - SmallDenseMap<const TreeEntry *, SmallVector<Value *>>; - auto GetScalarTy = [&](const TreeEntry *TE) { - Type *ScalarTy = TE->Scalars.front()->getType(); - auto It = MinBWs.find(TE); - if (It != MinBWs.end()) - ScalarTy = IntegerType::get(ScalarTy->getContext(), It->second.first); - return ScalarTy; - }; - // Lanes of the subtree scalars used by the surviving gather nodes, and the - // values to materialize in those gathers if the subtree is deleted. - auto FindDemandedElts = [&](TreeEntry *TE, ValuesToInsertTy &ValuesToInsert) { + // Check if all loads of gathered loads nodes are marked for deletion. In this + // case the whole gathered loads subtree must be deleted. + // Also, try to account for extracts, which might be required, if only part of + // gathered load must be vectorized. Keep partially vectorized nodes, if + // extracts are cheaper than gathers. + for (TreeEntry *TE : GatheredLoadsNodes) { + if (DeletedNodes.contains(TE) || TransformedToGatherNodes.contains(TE)) + continue; + GatheredLoadsToDelete.insert(TE); APInt DemandedElts = APInt::getZero(TE->getVectorFactor()); + // All loads are removed from gathered? Need to delete the subtree. + SmallDenseMap<const TreeEntry *, SmallVector<Value *>> ValuesToInsert; for (Value *V : TE->Scalars) { unsigned Pos = TE->findLaneForValue(V); for (const TreeEntry *BVE : ValueToGatherNodes.lookup(V)) { @@ -19777,52 +19679,34 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals, ValuesToInsert.try_emplace(BVE).first->second.push_back(V); } } - return DemandedElts; - }; - // Cost of materializing the values directly in the surviving gather nodes - // that use them. - auto GetGatherInsertCost = [&](Type *ScalarTy, - const ValuesToInsertTy &ValuesToInsert) { - InstructionCost BVCost = 0; - for (const auto &[BVE, Values] : ValuesToInsert) { - APInt BVDemandedElts = APInt::getZero(BVE->getVectorFactor()); - SmallVector<Value *> BVValues(BVE->getVectorFactor(), - PoisonValue::get(ScalarTy)); - for (Value *V : Values) { - unsigned Pos = BVE->findLaneForValue(V); - BVValues[Pos] = V; - BVDemandedElts.setBit(Pos); - } - BVCost += ::getScalarizationOverhead( - *TTI, ScalarTy, - cast<VectorType>(getWidenedType(ScalarTy, BVE->getVectorFactor())), - BVDemandedElts, /*Insert=*/true, /*Extract=*/false, CostKind, - BVDemandedElts.isAllOnes(), BVValues); - } - return BVCost; - }; - // Check if all loads of gathered loads nodes are marked for deletion. In this - // case the whole gathered loads subtree must be deleted. - // Also, try to account for extracts, which might be required, if only part of - // gathered load must be vectorized. Keep partially vectorized nodes, if - // extracts are cheaper than gathers. - for (TreeEntry *TE : GatheredLoadsNodes) { - if (DeletedNodes.contains(TE) || TransformedToGatherNodes.contains(TE)) - continue; - SubtreesToDelete.insert(TE); - // All loads are removed from gathered? Need to delete the subtree. - ValuesToInsertTy ValuesToInsert; - APInt DemandedElts = FindDemandedElts(TE, ValuesToInsert); if (!DemandedElts.isZero()) { - Type *ScalarTy = GetScalarTy(TE); + Type *ScalarTy = TE->Scalars.front()->getType(); + auto It = MinBWs.find(TE); + if (It != MinBWs.end()) + ScalarTy = IntegerType::get(ScalarTy->getContext(), It->second.first); auto *VecTy = getWidenedType(ScalarTy, TE->getVectorFactor()); InstructionCost ExtractsCost = ::getScalarizationOverhead( *TTI, ScalarTy, cast<VectorType>(VecTy), DemandedElts, /*Insert=*/false, /*Extract=*/true, CostKind); - InstructionCost BVCost = GetGatherInsertCost(ScalarTy, ValuesToInsert); + InstructionCost BVCost = 0; + for (const auto &[BVE, Values] : ValuesToInsert) { + APInt BVDemandedElts = APInt::getZero(BVE->getVectorFactor()); + SmallVector<Value *> BVValues(BVE->getVectorFactor(), + PoisonValue::get(ScalarTy)); + for (Value *V : Values) { + unsigned Pos = BVE->findLaneForValue(V); + BVValues[Pos] = V; + BVDemandedElts.setBit(Pos); + } + auto *BVVecTy = getWidenedType(ScalarTy, BVE->getVectorFactor()); + BVCost += ::getScalarizationOverhead( + *TTI, ScalarTy, cast<VectorType>(BVVecTy), BVDemandedElts, + /*Insert=*/true, /*Extract=*/false, CostKind, + BVDemandedElts.isAllOnes(), BVValues); + } if (ExtractsCost < BVCost) { LoadsExtractsCost += ExtractsCost; - SubtreesToDelete.erase(TE); + GatheredLoadsToDelete.erase(TE); continue; } LoadsExtractsCost += BVCost; @@ -19830,67 +19714,15 @@ BoUpSLP::calculateTreeCostAndTrimNonProfitable(ArrayRef<Value *> VectorizedVals, NodesCosts.erase(TE); } - // Check if all gather nodes that reuse the splat subtrees are marked for - // deletion. In this case the whole splat subtree must be deleted. If only - // some of the gathers are trimmed, keeping the subtree still costs its full - // price plus the extracts of the scalars used by the remaining scalar code, - // while the surviving gathers can materialize the splatted scalars - // directly. Drop the subtree if it does not pay off. - for (TreeEntry *TE : SplatGatheredScalarsRoots) { - if (DeletedNodes.contains(TE)) - continue; - ValuesToInsertTy ValuesToInsert; - APInt DemandedElts = FindDemandedElts(TE, ValuesToInsert); - if (!DemandedElts.isZero()) { - Type *ScalarTy = GetScalarTy(TE); - // Lanes of the subtree scalars still used by the remaining scalar code - // must be extracted if the subtree is kept. - APInt ExtractElts = APInt::getZero(TE->getVectorFactor()); - for (Value *V : TE->Scalars) { - if (!isa<Instruction>(V) || TE->isCopyableElement(V)) - continue; - // Too many users - the scalar is extracted anyway. - if (V->hasNUsesOrMore(UsesLimit) || any_of(V->users(), [&](User *U) { - return none_of(getTreeEntries(U), [&](const TreeEntry *UseTE) { - return !DeletedNodes.contains(UseTE) && - !TransformedToGatherNodes.contains(UseTE); - }); - })) - ExtractElts.setBit(TE->findLaneForValue(V)); - } - InstructionCost KeepCost = ::getScalarizationOverhead( - *TTI, ScalarTy, - cast<VectorType>(getWidenedType(ScalarTy, TE->getVectorFactor())), - ExtractElts, /*Insert=*/false, /*Extract=*/true, CostKind); - // Add the remaining cost of the subtree itself. The subtree node list - // covers the combined subnodes, which are not costed on their own. - auto GetLiveCost = [&](const TreeEntry *E) { - if (auto CostIt = NodesCosts.find(E); CostIt != NodesCosts.end()) - return CostIt->second; - return TransformedToGatherNodes.lookup(E); - }; - KeepCost += GetLiveCost(TE); - for (unsigned Idx : std::get<2>(SubtreeCosts[TE->Idx])) - if (!DeletedNodes.contains(VectorizableTree[Idx].get())) - KeepCost += GetLiveCost(VectorizableTree[Idx].get()); - InstructionCost DropCost = GetGatherInsertCost(ScalarTy, ValuesToInsert); - if (KeepCost <= DropCost) - continue; - } - // Not used by the surviving gathers or not profitable to keep. - SubtreesToDelete.insert(TE); - NodesCosts.erase(TE); - } - - // Deleted all subtrees rooted at gathered loads nodes or splat subtrees. + // Deleted all subtrees rooted at gathered loads nodes. for (std::unique_ptr<TreeEntry> &TE : VectorizableTree) { if (TE->UserTreeIndex && - SubtreesToDelete.contains(TE->UserTreeIndex.UserTE)) { + GatheredLoadsToDelete.contains(TE->UserTreeIndex.UserTE)) { DeletedNodes.insert(TE.get()); NodesCosts.erase(TE.get()); - SubtreesToDelete.insert(TE.get()); + GatheredLoadsToDelete.insert(TE.get()); } - if (SubtreesToDelete.contains(TE.get())) + if (GatheredLoadsToDelete.contains(TE.get())) DeletedNodes.insert(TE.get()); } @@ -25629,14 +25461,6 @@ Value *BoUpSLP::vectorizeTree( Builder.SetCurrentDebugLocation(Entry.second->getDebugLoc()); (void)vectorizeTree(Entry.first); } - // Emit the subtrees built for the splat gather nodes' unique scalars, so - // the splat gathers can be emitted as their broadcasts. They go before the - // gathered loads, which skip entries that already have a vector value. - for (TreeEntry *TE : SplatGatheredScalarsRoots) { - if (DeletedNodes.contains(TE) || TE->VectorizedValue) - continue; - (void)vectorizeTree(TE); - } // Emit gathered loads first to emit better code for the users of those // gathered loads. for (const std::unique_ptr<TreeEntry> &TE : VectorizableTree) { diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/loadi8.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/loadi8.ll index 4693cfafef019..f3e925f915c6c 100644 --- a/llvm/test/Transforms/SLPVectorizer/AArch64/loadi8.ll +++ b/llvm/test/Transforms/SLPVectorizer/AArch64/loadi8.ll @@ -9,12 +9,16 @@ target triple = "aarch64" define void @f_noalias(ptr noalias nocapture %dst, ptr noalias nocapture readonly %src, ptr noalias nocapture readonly %w) { ; CHECK-LABEL: @f_noalias( ; CHECK-NEXT: entry: -; CHECK-NEXT: [[TMP0:%.*]] = load <2 x i32>, ptr [[W:%.*]], align 16 +; CHECK-NEXT: [[TMP0:%.*]] = load i32, ptr [[W:%.*]], align 16 +; CHECK-NEXT: [[OFFSET:%.*]] = getelementptr inbounds [[STRUCT_WEIGHT_T:%.*]], ptr [[W]], i64 0, i32 1 +; CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr [[OFFSET]], align 4 ; CHECK-NEXT: [[TMP2:%.*]] = load <4 x i8>, ptr [[SRC:%.*]], align 1 ; CHECK-NEXT: [[TMP3:%.*]] = zext <4 x i8> [[TMP2]] to <4 x i32> -; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> poison, <4 x i32> zeroinitializer +; CHECK-NEXT: [[TMP4:%.*]] = insertelement <4 x i32> poison, i32 [[TMP0]], i64 0 +; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <4 x i32> [[TMP4]], <4 x i32> poison, <4 x i32> zeroinitializer ; CHECK-NEXT: [[TMP6:%.*]] = mul nsw <4 x i32> [[TMP5]], [[TMP3]] -; CHECK-NEXT: [[TMP8:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> poison, <4 x i32> <i32 1, i32 1, i32 1, i32 1> +; CHECK-NEXT: [[TMP7:%.*]] = insertelement <4 x i32> poison, i32 [[TMP1]], i64 0 +; CHECK-NEXT: [[TMP8:%.*]] = shufflevector <4 x i32> [[TMP7]], <4 x i32> poison, <4 x i32> zeroinitializer ; CHECK-NEXT: [[TMP9:%.*]] = add nsw <4 x i32> [[TMP6]], [[TMP8]] ; CHECK-NEXT: [[TMP10:%.*]] = icmp ult <4 x i32> [[TMP9]], splat (i32 256) ; CHECK-NEXT: [[TMP11:%.*]] = icmp sgt <4 x i32> [[TMP9]], zeroinitializer diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/matmul.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/matmul.ll index 0c1d5e2b2a7a2..28ac7624e2488 100644 --- a/llvm/test/Transforms/SLPVectorizer/AArch64/matmul.ll +++ b/llvm/test/Transforms/SLPVectorizer/AArch64/matmul.ll @@ -10,24 +10,32 @@ target datalayout = "e-m:e-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128" define void @wrap_mul4(ptr nocapture %Out, ptr nocapture readonly %A, ptr nocapture readonly %B) { ; CHECK-LABEL: @wrap_mul4( +; CHECK-NEXT: [[TEMP:%.*]] = load double, ptr [[A:%.*]], align 8 +; CHECK-NEXT: [[ARRAYIDX5_I:%.*]] = getelementptr inbounds [2 x double], ptr [[A]], i64 0, i64 1 +; CHECK-NEXT: [[TEMP2:%.*]] = load double, ptr [[ARRAYIDX5_I]], align 8 ; CHECK-NEXT: [[ARRAYIDX7_I:%.*]] = getelementptr inbounds [4 x double], ptr [[B:%.*]], i64 1, i64 0 -; CHECK-NEXT: [[ARRAYIDX47_I:%.*]] = getelementptr inbounds [2 x double], ptr [[A:%.*]], i64 1, i64 0 -; CHECK-NEXT: [[TMP2:%.*]] = load <2 x double>, ptr [[A]], align 8 +; CHECK-NEXT: [[ARRAYIDX47_I:%.*]] = getelementptr inbounds [2 x double], ptr [[A]], i64 1, i64 0 +; CHECK-NEXT: [[TEMP10:%.*]] = load double, ptr [[ARRAYIDX47_I]], align 8 +; CHECK-NEXT: [[ARRAYIDX52_I:%.*]] = getelementptr inbounds [2 x double], ptr [[A]], i64 1, i64 1 +; CHECK-NEXT: [[TEMP11:%.*]] = load double, ptr [[ARRAYIDX52_I]], align 8 ; CHECK-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[B]], align 8 -; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> poison, <4 x i32> zeroinitializer +; CHECK-NEXT: [[TMP2:%.*]] = insertelement <4 x double> poison, double [[TEMP]], i64 0 +; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <4 x double> [[TMP2]], <4 x double> poison, <4 x i32> zeroinitializer ; CHECK-NEXT: [[TMP4:%.*]] = fmul <4 x double> [[TMP3]], [[TMP1]] ; CHECK-NEXT: [[TMP5:%.*]] = load <4 x double>, ptr [[ARRAYIDX7_I]], align 8 -; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> poison, <4 x i32> <i32 1, i32 1, i32 1, i32 1> +; CHECK-NEXT: [[TMP6:%.*]] = insertelement <4 x double> poison, double [[TEMP2]], i64 0 +; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <4 x double> [[TMP6]], <4 x double> poison, <4 x i32> zeroinitializer ; CHECK-NEXT: [[TMP8:%.*]] = fmul <4 x double> [[TMP7]], [[TMP5]] ; CHECK-NEXT: [[TMP9:%.*]] = fadd <4 x double> [[TMP4]], [[TMP8]] -; CHECK-NEXT: [[RES_I_SROA_7_0_OUT2_I_SROA_IDX8:%.*]] = getelementptr inbounds double, ptr [[OUT:%.*]], i64 4 -; CHECK-NEXT: [[TMP10:%.*]] = load <2 x double>, ptr [[ARRAYIDX47_I]], align 8 -; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <2 x double> [[TMP10]], <2 x double> poison, <4 x i32> zeroinitializer +; CHECK-NEXT: store <4 x double> [[TMP9]], ptr [[OUT:%.*]], align 8 +; CHECK-NEXT: [[RES_I_SROA_7_0_OUT2_I_SROA_IDX8:%.*]] = getelementptr inbounds double, ptr [[OUT]], i64 4 +; CHECK-NEXT: [[TMP10:%.*]] = insertelement <4 x double> poison, double [[TEMP10]], i64 0 +; CHECK-NEXT: [[TMP11:%.*]] = shufflevector <4 x double> [[TMP10]], <4 x double> poison, <4 x i32> zeroinitializer ; CHECK-NEXT: [[TMP12:%.*]] = fmul <4 x double> [[TMP1]], [[TMP11]] -; CHECK-NEXT: [[TMP14:%.*]] = shufflevector <2 x double> [[TMP10]], <2 x double> poison, <4 x i32> <i32 1, i32 1, i32 1, i32 1> +; CHECK-NEXT: [[TMP13:%.*]] = insertelement <4 x double> poison, double [[TEMP11]], i64 0 +; CHECK-NEXT: [[TMP14:%.*]] = shufflevector <4 x double> [[TMP13]], <4 x double> poison, <4 x i32> zeroinitializer ; CHECK-NEXT: [[TMP15:%.*]] = fmul <4 x double> [[TMP5]], [[TMP14]] ; CHECK-NEXT: [[TMP16:%.*]] = fadd <4 x double> [[TMP12]], [[TMP15]] -; CHECK-NEXT: store <4 x double> [[TMP9]], ptr [[OUT]], align 8 ; CHECK-NEXT: store <4 x double> [[TMP16]], ptr [[RES_I_SROA_7_0_OUT2_I_SROA_IDX8]], align 8 ; CHECK-NEXT: ret void ; diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-loads.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-loads.ll index a85ab3f7cbb66..b6dc4735a0be4 100644 --- a/llvm/test/Transforms/SLPVectorizer/AArch64/splat-loads.ll +++ b/llvm/test/Transforms/SLPVectorizer/AArch64/splat-loads.ll @@ -8,11 +8,15 @@ target triple = "aarch64--linux-gnu" define void @splat_loads_double(ptr %array1, ptr %array2, ptr %ptrA, ptr %ptrB) { ; CHECK-LABEL: @splat_loads_double( ; CHECK-NEXT: entry: +; CHECK-NEXT: [[GEP_2_1:%.*]] = getelementptr inbounds double, ptr [[ARRAY2:%.*]], i64 1 +; CHECK-NEXT: [[LD_2_0:%.*]] = load double, ptr [[ARRAY2]], align 8 +; CHECK-NEXT: [[LD_2_1:%.*]] = load double, ptr [[GEP_2_1]], align 8 ; CHECK-NEXT: [[TMP0:%.*]] = load <2 x double>, ptr [[ARRAY1:%.*]], align 8 -; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[ARRAY2:%.*]], align 8 +; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x double> poison, double [[LD_2_0]], i64 0 ; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> zeroinitializer ; CHECK-NEXT: [[TMP3:%.*]] = fmul <2 x double> [[TMP0]], [[TMP2]] -; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> <i32 1, i32 1> +; CHECK-NEXT: [[TMP4:%.*]] = insertelement <2 x double> poison, double [[LD_2_1]], i64 0 +; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x double> [[TMP4]], <2 x double> poison, <2 x i32> zeroinitializer ; CHECK-NEXT: [[TMP6:%.*]] = fmul <2 x double> [[TMP0]], [[TMP5]] ; CHECK-NEXT: [[TMP7:%.*]] = fadd <2 x double> [[TMP3]], [[TMP6]] ; CHECK-NEXT: store <2 x double> [[TMP7]], ptr [[ARRAY1]], align 8 @@ -45,11 +49,15 @@ entry: define void @splat_loads_float(ptr %array1, ptr %array2, ptr %ptrA, ptr %ptrB) { ; CHECK-LABEL: @splat_loads_float( ; CHECK-NEXT: entry: +; CHECK-NEXT: [[GEP_2_1:%.*]] = getelementptr inbounds float, ptr [[ARRAY2:%.*]], i64 1 +; CHECK-NEXT: [[LD_2_0:%.*]] = load float, ptr [[ARRAY2]], align 8 +; CHECK-NEXT: [[LD_2_1:%.*]] = load float, ptr [[GEP_2_1]], align 8 ; CHECK-NEXT: [[TMP0:%.*]] = load <2 x float>, ptr [[ARRAY1:%.*]], align 8 -; CHECK-NEXT: [[TMP1:%.*]] = load <2 x float>, ptr [[ARRAY2:%.*]], align 8 +; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x float> poison, float [[LD_2_0]], i64 0 ; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x float> [[TMP1]], <2 x float> poison, <2 x i32> zeroinitializer ; CHECK-NEXT: [[TMP3:%.*]] = fmul <2 x float> [[TMP0]], [[TMP2]] -; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x float> [[TMP1]], <2 x float> poison, <2 x i32> <i32 1, i32 1> +; CHECK-NEXT: [[TMP4:%.*]] = insertelement <2 x float> poison, float [[LD_2_1]], i64 0 +; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x float> [[TMP4]], <2 x float> poison, <2 x i32> zeroinitializer ; CHECK-NEXT: [[TMP6:%.*]] = fmul <2 x float> [[TMP0]], [[TMP5]] ; CHECK-NEXT: [[TMP7:%.*]] = fadd <2 x float> [[TMP3]], [[TMP6]] ; CHECK-NEXT: store <2 x float> [[TMP7]], ptr [[ARRAY1]], align 4 @@ -82,11 +90,15 @@ entry: define void @splat_loads_i64(ptr %array1, ptr %array2, ptr %ptrA, ptr %ptrB) { ; CHECK-LABEL: @splat_loads_i64( ; CHECK-NEXT: entry: +; CHECK-NEXT: [[GEP_2_2:%.*]] = getelementptr inbounds i64, ptr [[ARRAY3:%.*]], i64 1 +; CHECK-NEXT: [[LD_2_2:%.*]] = load i64, ptr [[ARRAY3]], align 8 +; CHECK-NEXT: [[LD_2_3:%.*]] = load i64, ptr [[GEP_2_2]], align 8 ; CHECK-NEXT: [[TMP0:%.*]] = load <2 x i64>, ptr [[ARRAY1:%.*]], align 8 -; CHECK-NEXT: [[TMP1:%.*]] = load <2 x i64>, ptr [[ARRAY2:%.*]], align 8 +; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x i64> poison, i64 [[LD_2_2]], i64 0 ; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x i64> [[TMP1]], <2 x i64> poison, <2 x i32> zeroinitializer ; CHECK-NEXT: [[TMP3:%.*]] = or <2 x i64> [[TMP0]], [[TMP2]] -; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x i64> [[TMP1]], <2 x i64> poison, <2 x i32> <i32 1, i32 1> +; CHECK-NEXT: [[TMP4:%.*]] = insertelement <2 x i64> poison, i64 [[LD_2_3]], i64 0 +; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x i64> [[TMP4]], <2 x i64> poison, <2 x i32> zeroinitializer ; CHECK-NEXT: [[TMP6:%.*]] = or <2 x i64> [[TMP0]], [[TMP5]] ; CHECK-NEXT: [[TMP7:%.*]] = add <2 x i64> [[TMP3]], [[TMP6]] ; CHECK-NEXT: store <2 x i64> [[TMP7]], ptr [[ARRAY1]], align 8 @@ -119,11 +131,15 @@ entry: define void @splat_loads_i32(ptr %array1, ptr %array2, ptr %ptrA, ptr %ptrB) { ; CHECK-LABEL: @splat_loads_i32( ; CHECK-NEXT: entry: +; CHECK-NEXT: [[GEP_2_2:%.*]] = getelementptr inbounds i32, ptr [[ARRAY3:%.*]], i64 1 +; CHECK-NEXT: [[LD_2_2:%.*]] = load i32, ptr [[ARRAY3]], align 8 +; CHECK-NEXT: [[LD_2_3:%.*]] = load i32, ptr [[GEP_2_2]], align 8 ; CHECK-NEXT: [[TMP0:%.*]] = load <2 x i32>, ptr [[ARRAY1:%.*]], align 8 -; CHECK-NEXT: [[TMP1:%.*]] = load <2 x i32>, ptr [[ARRAY2:%.*]], align 8 +; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x i32> poison, i32 [[LD_2_2]], i64 0 ; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> zeroinitializer ; CHECK-NEXT: [[TMP3:%.*]] = or <2 x i32> [[TMP0]], [[TMP2]] -; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> <i32 1, i32 1> +; CHECK-NEXT: [[TMP4:%.*]] = insertelement <2 x i32> poison, i32 [[LD_2_3]], i64 0 +; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x i32> [[TMP4]], <2 x i32> poison, <2 x i32> zeroinitializer ; CHECK-NEXT: [[TMP6:%.*]] = or <2 x i32> [[TMP0]], [[TMP5]] ; CHECK-NEXT: [[TMP7:%.*]] = add <2 x i32> [[TMP3]], [[TMP6]] ; CHECK-NEXT: store <2 x i32> [[TMP7]], ptr [[ARRAY1]], align 4 diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-extracts.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-extracts.ll index b8865f42a2dd0..7de43dab6e480 100644 --- a/llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-extracts.ll +++ b/llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-extracts.ll @@ -11,29 +11,31 @@ define void @splat_trim(ptr %A, ptr %C, ptr %D, ptr %F, ptr %S, ptr %E, ptr %B) ; CHECK-NEXT: [[D0:%.*]] = load i64, ptr [[D]], align 8 ; CHECK-NEXT: [[E0:%.*]] = load i64, ptr [[E]], align 8 ; CHECK-NEXT: [[F0:%.*]] = load i64, ptr [[F]], align 8 -; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x i64> poison, i64 [[E0]], i64 0 +; CHECK-NEXT: [[TMP0:%.*]] = insertelement <2 x i64> poison, i64 [[A0]], i64 0 ; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x i64> [[TMP0]], i64 [[B0]], i64 1 ; CHECK-NEXT: [[TMP2:%.*]] = add <2 x i64> splat (i64 1), [[TMP1]] -; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <2 x i64> [[TMP2]], <2 x i64> poison, <4 x i32> <i32 0, i32 1, i32 0, i32 0> -; CHECK-NEXT: [[TMP4:%.*]] = insertelement <2 x i64> poison, i64 [[A0]], i64 0 +; CHECK-NEXT: [[TMP7:%.*]] = extractelement <2 x i64> [[TMP2]], i64 1 +; CHECK-NEXT: [[TMP8:%.*]] = extractelement <2 x i64> [[TMP2]], i64 0 +; CHECK-NEXT: [[TMP4:%.*]] = insertelement <2 x i64> poison, i64 [[E0]], i64 0 ; CHECK-NEXT: [[TMP5:%.*]] = insertelement <2 x i64> [[TMP4]], i64 [[B0]], i64 1 ; CHECK-NEXT: [[TMP6:%.*]] = add <2 x i64> splat (i64 1), [[TMP5]] -; CHECK-NEXT: [[TMP7:%.*]] = extractelement <2 x i64> [[TMP6]], i64 1 -; CHECK-NEXT: [[TMP8:%.*]] = extractelement <2 x i64> [[TMP6]], i64 0 +; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <2 x i64> [[TMP6]], <2 x i64> poison, <4 x i32> <i32 0, i32 1, i32 0, i32 0> ; CHECK-NEXT: [[AS2:%.*]] = add i64 [[TMP8]], [[TMP7]] ; CHECK-NEXT: [[AS3:%.*]] = sub i64 [[TMP8]], [[TMP7]] ; CHECK-NEXT: [[SH2:%.*]] = lshr i64 [[AS2]], [[E0]] ; CHECK-NEXT: [[SH3:%.*]] = shl i64 [[AS3]], [[F0]] -; CHECK-NEXT: [[TMP9:%.*]] = shufflevector <2 x i64> [[TMP6]], <2 x i64> poison, <2 x i32> zeroinitializer -; CHECK-NEXT: [[TMP10:%.*]] = shufflevector <2 x i64> [[TMP6]], <2 x i64> poison, <2 x i32> <i32 1, i32 1> +; CHECK-NEXT: [[AS13:%.*]] = sub i64 [[SH2]], [[E0]] +; CHECK-NEXT: [[AS12:%.*]] = add i64 [[SH3]], [[F0]] +; CHECK-NEXT: [[TMP23:%.*]] = insertelement <2 x i64> poison, i64 [[TMP8]], i64 0 +; CHECK-NEXT: [[TMP9:%.*]] = shufflevector <2 x i64> [[TMP23]], <2 x i64> poison, <2 x i32> zeroinitializer +; CHECK-NEXT: [[TMP24:%.*]] = insertelement <2 x i64> poison, i64 [[TMP7]], i64 0 +; CHECK-NEXT: [[TMP10:%.*]] = shufflevector <2 x i64> [[TMP24]], <2 x i64> poison, <2 x i32> zeroinitializer ; CHECK-NEXT: [[TMP11:%.*]] = add <2 x i64> [[TMP9]], [[TMP10]] ; CHECK-NEXT: [[TMP12:%.*]] = insertelement <2 x i64> poison, i64 [[C0]], i64 0 ; CHECK-NEXT: [[TMP13:%.*]] = insertelement <2 x i64> [[TMP12]], i64 [[D0]], i64 1 ; CHECK-NEXT: [[TMP14:%.*]] = shl <2 x i64> [[TMP11]], [[TMP13]] ; CHECK-NEXT: [[TMP15:%.*]] = lshr <2 x i64> [[TMP11]], [[TMP13]] ; CHECK-NEXT: [[TMP16:%.*]] = shufflevector <2 x i64> [[TMP14]], <2 x i64> [[TMP15]], <2 x i32> <i32 0, i32 3> -; CHECK-NEXT: [[AS13:%.*]] = sub i64 [[SH2]], [[E0]] -; CHECK-NEXT: [[AS12:%.*]] = add i64 [[SH3]], [[F0]] ; CHECK-NEXT: [[TMP17:%.*]] = add <2 x i64> [[TMP16]], [[TMP13]] ; CHECK-NEXT: [[TMP18:%.*]] = shufflevector <2 x i64> [[TMP17]], <2 x i64> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison> ; CHECK-NEXT: [[TMP19:%.*]] = insertelement <4 x i64> [[TMP18]], i64 [[AS12]], i64 2 diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-trim-combined.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-trim-combined.ll index d449a03586868..9724750bc7f0e 100644 --- a/llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-trim-combined.ll +++ b/llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-trim-combined.ll @@ -11,12 +11,12 @@ define void @splat_trim_combined(ptr %A, ptr %B, ptr %C, ptr %D, ptr %E, ptr %F, ; CHECK-NEXT: [[D0:%.*]] = load i64, ptr [[D]], align 8 ; CHECK-NEXT: [[E0:%.*]] = load i64, ptr [[E]], align 8 ; CHECK-NEXT: [[F0:%.*]] = load i64, ptr [[F]], align 8 -; CHECK-NEXT: [[C3:%.*]] = icmp sgt i64 [[E0]], 42 -; CHECK-NEXT: [[C2:%.*]] = icmp sgt i64 [[C0]], 42 ; CHECK-NEXT: [[C1:%.*]] = icmp sgt i64 [[A0]], 42 -; CHECK-NEXT: [[V3:%.*]] = select i1 [[C3]], i64 [[E0]], i64 42 -; CHECK-NEXT: [[V2:%.*]] = select i1 [[C2]], i64 [[C0]], i64 42 ; CHECK-NEXT: [[V1:%.*]] = select i1 [[C1]], i64 [[A0]], i64 42 +; CHECK-NEXT: [[C2:%.*]] = icmp sgt i64 [[C0]], 42 +; CHECK-NEXT: [[V2:%.*]] = select i1 [[C2]], i64 [[C0]], i64 42 +; CHECK-NEXT: [[C3:%.*]] = icmp sgt i64 [[E0]], 42 +; CHECK-NEXT: [[V3:%.*]] = select i1 [[C3]], i64 [[E0]], i64 42 ; CHECK-NEXT: [[AS2:%.*]] = add i64 [[V1]], [[V2]] ; CHECK-NEXT: [[AS3:%.*]] = sub i64 [[V1]], [[V2]] ; CHECK-NEXT: [[AS1:%.*]] = add i64 [[V1]], [[V2]] diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-trim.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-trim.ll index ee8eb17575654..8263b9d8e51a8 100644 --- a/llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-trim.ll +++ b/llvm/test/Transforms/SLPVectorizer/RISCV/splat-gather-trim.ll @@ -12,8 +12,8 @@ define void @splat_trim(ptr %A, ptr %B, ptr %C, ptr %D, ptr %E, ptr %F, ptr %S) ; CHECK-NEXT: [[D0:%.*]] = load i64, ptr [[D]], align 8 ; CHECK-NEXT: [[E0:%.*]] = load i64, ptr [[E]], align 8 ; CHECK-NEXT: [[F0:%.*]] = load i64, ptr [[F]], align 8 -; CHECK-NEXT: [[V2:%.*]] = add i64 1, [[B0]] ; CHECK-NEXT: [[V1:%.*]] = add i64 1, [[A0]] +; CHECK-NEXT: [[V2:%.*]] = add i64 1, [[B0]] ; CHECK-NEXT: [[AS2:%.*]] = add i64 [[V1]], [[V2]] ; CHECK-NEXT: [[AS3:%.*]] = sub i64 [[V1]], [[V2]] ; CHECK-NEXT: [[AS1:%.*]] = add i64 [[V1]], [[V2]] @@ -44,8 +44,8 @@ define void @splat_trim(ptr %A, ptr %B, ptr %C, ptr %D, ptr %E, ptr %F, ptr %S) ; THRESH-NEXT: [[D0:%.*]] = load i64, ptr [[D]], align 8 ; THRESH-NEXT: [[E0:%.*]] = load i64, ptr [[E]], align 8 ; THRESH-NEXT: [[F0:%.*]] = load i64, ptr [[F]], align 8 -; THRESH-NEXT: [[V2:%.*]] = add i64 1, [[B0]] ; THRESH-NEXT: [[V1:%.*]] = add i64 1, [[A0]] +; THRESH-NEXT: [[V2:%.*]] = add i64 1, [[B0]] ; THRESH-NEXT: [[AS2:%.*]] = add i64 [[V1]], [[V2]] ; THRESH-NEXT: [[AS3:%.*]] = sub i64 [[V1]], [[V2]] ; THRESH-NEXT: [[AS1:%.*]] = add i64 [[V1]], [[V2]] @@ -124,9 +124,9 @@ define void @splat_trim_partial(ptr %A, ptr %B, ptr %C, ptr %D, ptr %E, ptr %F, ; CHECK-NEXT: [[D0:%.*]] = load i64, ptr [[D]], align 8 ; CHECK-NEXT: [[E0:%.*]] = load i64, ptr [[E]], align 8 ; CHECK-NEXT: [[F0:%.*]] = load i64, ptr [[F]], align 8 -; CHECK-NEXT: [[V3:%.*]] = add i64 1, [[E0]] -; CHECK-NEXT: [[V2:%.*]] = add i64 1, [[B0]] ; CHECK-NEXT: [[V1:%.*]] = add i64 1, [[A0]] +; CHECK-NEXT: [[V2:%.*]] = add i64 1, [[B0]] +; CHECK-NEXT: [[V3:%.*]] = add i64 1, [[E0]] ; CHECK-NEXT: [[AS0:%.*]] = add i64 [[V1]], [[V2]] ; CHECK-NEXT: [[AS3:%.*]] = sub i64 [[V1]], [[V2]] ; CHECK-NEXT: [[AS1:%.*]] = add i64 [[V1]], [[V2]] @@ -159,9 +159,9 @@ define void @splat_trim_partial(ptr %A, ptr %B, ptr %C, ptr %D, ptr %E, ptr %F, ; THRESH-NEXT: [[D0:%.*]] = load i64, ptr [[D]], align 8 ; THRESH-NEXT: [[E0:%.*]] = load i64, ptr [[E]], align 8 ; THRESH-NEXT: [[F0:%.*]] = load i64, ptr [[F]], align 8 -; THRESH-NEXT: [[V3:%.*]] = add i64 1, [[E0]] -; THRESH-NEXT: [[V2:%.*]] = add i64 1, [[B0]] ; THRESH-NEXT: [[TMP4:%.*]] = add i64 1, [[A0]] +; THRESH-NEXT: [[V2:%.*]] = add i64 1, [[B0]] +; THRESH-NEXT: [[V3:%.*]] = add i64 1, [[E0]] ; THRESH-NEXT: [[AS2:%.*]] = add i64 [[TMP4]], [[V2]] ; THRESH-NEXT: [[AS3:%.*]] = sub i64 [[TMP4]], [[V2]] ; THRESH-NEXT: [[AS1:%.*]] = add i64 [[TMP4]], [[V2]] diff --git a/llvm/test/Transforms/SLPVectorizer/X86/broadcast.ll b/llvm/test/Transforms/SLPVectorizer/X86/broadcast.ll index d25f4654d1824..c16f8d8948320 100644 --- a/llvm/test/Transforms/SLPVectorizer/X86/broadcast.ll +++ b/llvm/test/Transforms/SLPVectorizer/X86/broadcast.ll @@ -18,8 +18,10 @@ define void @bcast_vals(ptr %A, ptr %B, ptr %S) { ; CHECK-NEXT: [[TMP5:%.*]] = insertelement <2 x i64> poison, i64 [[A0]], i64 0 ; CHECK-NEXT: [[TMP6:%.*]] = insertelement <2 x i64> [[TMP5]], i64 [[B0]], i64 1 ; CHECK-NEXT: [[TMP7:%.*]] = sub <2 x i64> [[TMP6]], splat (i64 1) -; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <2 x i64> [[TMP7]], <2 x i64> poison, <4 x i32> zeroinitializer -; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <2 x i64> [[TMP7]], <2 x i64> poison, <4 x i32> <i32 1, i32 1, i32 1, i32 1> +; CHECK-NEXT: [[TMP0:%.*]] = shufflevector <2 x i64> [[TMP7]], <2 x i64> poison, <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison> +; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <4 x i64> [[TMP0]], <4 x i64> poison, <4 x i32> zeroinitializer +; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x i64> [[TMP7]], <2 x i64> poison, <4 x i32> <i32 1, i32 poison, i32 poison, i32 poison> +; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <4 x i64> [[TMP2]], <4 x i64> poison, <4 x i32> zeroinitializer ; CHECK-NEXT: [[TMP4:%.*]] = add <4 x i64> [[TMP1]], [[TMP3]] ; CHECK-NEXT: store <4 x i64> [[TMP4]], ptr [[S:%.*]], align 8 ; CHECK-NEXT: ret void diff --git a/llvm/test/Transforms/SLPVectorizer/X86/lookahead.ll b/llvm/test/Transforms/SLPVectorizer/X86/lookahead.ll index e96c0c32272a7..ba15324f30e51 100644 --- a/llvm/test/Transforms/SLPVectorizer/X86/lookahead.ll +++ b/llvm/test/Transforms/SLPVectorizer/X86/lookahead.ll @@ -387,12 +387,16 @@ define void @lookahead_crash(ptr %A, ptr %S, ptr %Arg0) { ; This checks that we choose to group consecutive extracts from the same vectors. define void @ChecksExtractScores(ptr %storeArray, ptr %array, ptr %vecPtr1, ptr %vecPtr2) { ; CHECK-LABEL: @ChecksExtractScores( +; CHECK-NEXT: [[IDX1:%.*]] = getelementptr inbounds double, ptr [[ARRAY:%.*]], i64 1 +; CHECK-NEXT: [[LOADA0:%.*]] = load double, ptr [[ARRAY]], align 4 +; CHECK-NEXT: [[LOADA1:%.*]] = load double, ptr [[IDX1]], align 4 ; CHECK-NEXT: [[LOADVEC:%.*]] = load <2 x double>, ptr [[VECPTR1:%.*]], align 4 ; CHECK-NEXT: [[LOADVEC2:%.*]] = load <2 x double>, ptr [[VECPTR2:%.*]], align 4 -; CHECK-NEXT: [[TMP1:%.*]] = load <2 x double>, ptr [[ARRAY:%.*]], align 4 +; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x double> poison, double [[LOADA0]], i64 0 ; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> zeroinitializer ; CHECK-NEXT: [[TMP3:%.*]] = fmul <2 x double> [[LOADVEC]], [[TMP2]] -; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> <i32 1, i32 1> +; CHECK-NEXT: [[TMP4:%.*]] = insertelement <2 x double> poison, double [[LOADA1]], i64 0 +; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <2 x double> [[TMP4]], <2 x double> poison, <2 x i32> zeroinitializer ; CHECK-NEXT: [[TMP6:%.*]] = fmul <2 x double> [[LOADVEC2]], [[TMP5]] ; CHECK-NEXT: [[TMP7:%.*]] = fadd <2 x double> [[TMP3]], [[TMP6]] ; CHECK-NEXT: store <2 x double> [[TMP7]], ptr [[STOREARRAY:%.*]], align 8 @@ -531,16 +535,20 @@ define void @ChecksExtractScores_ diff erent_vectors(ptr %storeArray, ptr %array, ; SSE-NEXT: ret void ; ; AVX-LABEL: @ChecksExtractScores_ diff erent_vectors( +; AVX-NEXT: [[IDX1:%.*]] = getelementptr inbounds double, ptr [[ARRAY:%.*]], i64 1 +; AVX-NEXT: [[LOADA0:%.*]] = load double, ptr [[ARRAY]], align 4 +; AVX-NEXT: [[LOADA1:%.*]] = load double, ptr [[IDX1]], align 4 ; AVX-NEXT: [[LOADVEC:%.*]] = load <2 x double>, ptr [[VECPTR1:%.*]], align 4 ; AVX-NEXT: [[LOADVEC2:%.*]] = load <2 x double>, ptr [[VECPTR2:%.*]], align 4 ; AVX-NEXT: [[LOADVEC3:%.*]] = load <2 x double>, ptr [[VECPTR3:%.*]], align 4 ; AVX-NEXT: [[LOADVEC4:%.*]] = load <2 x double>, ptr [[VECPTR4:%.*]], align 4 -; AVX-NEXT: [[TMP2:%.*]] = load <2 x double>, ptr [[ARRAY:%.*]], align 4 ; AVX-NEXT: [[TMP1:%.*]] = shufflevector <2 x double> [[LOADVEC]], <2 x double> [[LOADVEC2]], <2 x i32> <i32 0, i32 3> +; AVX-NEXT: [[TMP2:%.*]] = insertelement <2 x double> poison, double [[LOADA0]], i64 0 ; AVX-NEXT: [[TMP3:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> poison, <2 x i32> zeroinitializer ; AVX-NEXT: [[TMP4:%.*]] = fmul <2 x double> [[TMP1]], [[TMP3]] ; AVX-NEXT: [[TMP5:%.*]] = shufflevector <2 x double> [[LOADVEC3]], <2 x double> [[LOADVEC4]], <2 x i32> <i32 0, i32 3> -; AVX-NEXT: [[TMP7:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> poison, <2 x i32> <i32 1, i32 1> +; AVX-NEXT: [[TMP6:%.*]] = insertelement <2 x double> poison, double [[LOADA1]], i64 0 +; AVX-NEXT: [[TMP7:%.*]] = shufflevector <2 x double> [[TMP6]], <2 x double> poison, <2 x i32> zeroinitializer ; AVX-NEXT: [[TMP8:%.*]] = fmul <2 x double> [[TMP5]], [[TMP7]] ; AVX-NEXT: [[TMP9:%.*]] = fadd <2 x double> [[TMP4]], [[TMP8]] ; AVX-NEXT: store <2 x double> [[TMP9]], ptr [[STOREARRAY:%.*]], align 8 diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reorder_phi.ll b/llvm/test/Transforms/SLPVectorizer/X86/reorder_phi.ll index 4a71954781342..13e7bee0cdd5d 100644 --- a/llvm/test/Transforms/SLPVectorizer/X86/reorder_phi.ll +++ b/llvm/test/Transforms/SLPVectorizer/X86/reorder_phi.ll @@ -13,11 +13,15 @@ define void @foo (ptr %A, ptr %B, ptr %Result) { ; CHECK-NEXT: [[TMP2:%.*]] = phi <2 x float> [ zeroinitializer, [[ENTRY]] ], [ [[TMP20:%.*]], [[LOOP]] ] ; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds [[STRUCT_COMPLEX:%.*]], ptr [[A:%.*]], i64 [[TMP1]], i32 0 ; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds [[STRUCT_COMPLEX]], ptr [[B:%.*]], i64 [[TMP1]], i32 0 +; CHECK-NEXT: [[TMP5:%.*]] = load float, ptr [[TMP4]], align 4 +; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds [[STRUCT_COMPLEX]], ptr [[B]], i64 [[TMP1]], i32 1 +; CHECK-NEXT: [[TMP7:%.*]] = load float, ptr [[TMP6]], align 4 ; CHECK-NEXT: [[TMP8:%.*]] = load <2 x float>, ptr [[TMP3]], align 4 -; CHECK-NEXT: [[TMP9:%.*]] = load <2 x float>, ptr [[TMP4]], align 4 +; CHECK-NEXT: [[TMP9:%.*]] = insertelement <2 x float> poison, float [[TMP5]], i64 0 ; CHECK-NEXT: [[TMP10:%.*]] = shufflevector <2 x float> [[TMP9]], <2 x float> poison, <2 x i32> zeroinitializer ; CHECK-NEXT: [[TMP11:%.*]] = fmul <2 x float> [[TMP8]], [[TMP10]] -; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <2 x float> [[TMP9]], <2 x float> poison, <2 x i32> <i32 1, i32 1> +; CHECK-NEXT: [[TMP12:%.*]] = insertelement <2 x float> poison, float [[TMP7]], i64 0 +; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <2 x float> [[TMP12]], <2 x float> poison, <2 x i32> zeroinitializer ; CHECK-NEXT: [[TMP14:%.*]] = fmul <2 x float> [[TMP8]], [[TMP13]] ; CHECK-NEXT: [[TMP15:%.*]] = shufflevector <2 x float> [[TMP14]], <2 x float> poison, <2 x i32> <i32 1, i32 0> ; CHECK-NEXT: [[TMP16:%.*]] = fsub <2 x float> [[TMP11]], [[TMP15]] diff --git a/llvm/test/Transforms/SLPVectorizer/X86/splat-gather-operands.ll b/llvm/test/Transforms/SLPVectorizer/X86/splat-gather-operands.ll index 4c8babbe46992..b16ae4839d16c 100644 --- a/llvm/test/Transforms/SLPVectorizer/X86/splat-gather-operands.ll +++ b/llvm/test/Transforms/SLPVectorizer/X86/splat-gather-operands.ll @@ -14,40 +14,108 @@ define i32 @test(ptr noalias %pix1, i32 %i_pix1, ptr noalias %pix2, i32 %i_pix2, ; CHECK-NEXT: [[I1_I64_0:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[NEXTIVLOOP_1247:%.*]], [[LOOP_1247]] ] ; CHECK-NEXT: [[I:%.*]] = mul i64 [[I1_I64_0]], [[IDX_EXT_I]] ; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i8, ptr [[PIX1:%.*]], i64 [[I]] +; CHECK-NEXT: [[LOAD00:%.*]] = load i8, ptr [[GEP]], align 1 ; CHECK-NEXT: [[I1:%.*]] = mul i64 [[I1_I64_0]], [[IDX_EXT63_I]] ; CHECK-NEXT: [[GEP013:%.*]] = getelementptr inbounds i8, ptr [[PIX2:%.*]], i64 [[I1]] +; CHECK-NEXT: [[LOAD014:%.*]] = load i8, ptr [[GEP013]], align 1 ; CHECK-NEXT: [[I2:%.*]] = add i64 [[I]], 4 ; CHECK-NEXT: [[GEP015:%.*]] = getelementptr inbounds i8, ptr [[PIX1]], i64 [[I2]] +; CHECK-NEXT: [[LOAD016:%.*]] = load i8, ptr [[GEP015]], align 1 ; CHECK-NEXT: [[I3:%.*]] = add i64 [[I1]], 4 ; CHECK-NEXT: [[GEP017:%.*]] = getelementptr inbounds i8, ptr [[PIX2]], i64 [[I3]] +; CHECK-NEXT: [[LOAD018:%.*]] = load i8, ptr [[GEP017]], align 1 +; CHECK-NEXT: [[I4:%.*]] = add i64 [[I]], 1 +; CHECK-NEXT: [[GEP019:%.*]] = getelementptr inbounds i8, ptr [[PIX1]], i64 [[I4]] +; CHECK-NEXT: [[LOAD020:%.*]] = load i8, ptr [[GEP019]], align 1 +; CHECK-NEXT: [[I5:%.*]] = add i64 [[I1]], 1 +; CHECK-NEXT: [[GEP021:%.*]] = getelementptr inbounds i8, ptr [[PIX2]], i64 [[I5]] +; CHECK-NEXT: [[LOAD022:%.*]] = load i8, ptr [[GEP021]], align 1 +; CHECK-NEXT: [[I6:%.*]] = add i64 [[I]], 5 +; CHECK-NEXT: [[GEP023:%.*]] = getelementptr inbounds i8, ptr [[PIX1]], i64 [[I6]] +; CHECK-NEXT: [[LOAD024:%.*]] = load i8, ptr [[GEP023]], align 1 +; CHECK-NEXT: [[I7:%.*]] = add i64 [[I1]], 5 +; CHECK-NEXT: [[GEP025:%.*]] = getelementptr inbounds i8, ptr [[PIX2]], i64 [[I7]] +; CHECK-NEXT: [[LOAD026:%.*]] = load i8, ptr [[GEP025]], align 1 +; CHECK-NEXT: [[I8:%.*]] = add i64 [[I]], 2 +; CHECK-NEXT: [[GEP027:%.*]] = getelementptr inbounds i8, ptr [[PIX1]], i64 [[I8]] +; CHECK-NEXT: [[LOAD028:%.*]] = load i8, ptr [[GEP027]], align 1 +; CHECK-NEXT: [[I9:%.*]] = add i64 [[I1]], 2 +; CHECK-NEXT: [[GEP029:%.*]] = getelementptr inbounds i8, ptr [[PIX2]], i64 [[I9]] +; CHECK-NEXT: [[LOAD030:%.*]] = load i8, ptr [[GEP029]], align 1 +; CHECK-NEXT: [[I10:%.*]] = add i64 [[I]], 6 +; CHECK-NEXT: [[GEP031:%.*]] = getelementptr inbounds i8, ptr [[PIX1]], i64 [[I10]] +; CHECK-NEXT: [[LOAD032:%.*]] = load i8, ptr [[GEP031]], align 1 +; CHECK-NEXT: [[I11:%.*]] = add i64 [[I1]], 6 +; CHECK-NEXT: [[GEP033:%.*]] = getelementptr inbounds i8, ptr [[PIX2]], i64 [[I11]] +; CHECK-NEXT: [[LOAD034:%.*]] = load i8, ptr [[GEP033]], align 1 +; CHECK-NEXT: [[I12:%.*]] = add i64 [[I]], 3 +; CHECK-NEXT: [[GEP035:%.*]] = getelementptr inbounds i8, ptr [[PIX1]], i64 [[I12]] +; CHECK-NEXT: [[LOAD036:%.*]] = load i8, ptr [[GEP035]], align 1 +; CHECK-NEXT: [[I13:%.*]] = add i64 [[I1]], 3 +; CHECK-NEXT: [[GEP037:%.*]] = getelementptr inbounds i8, ptr [[PIX2]], i64 [[I13]] +; CHECK-NEXT: [[LOAD038:%.*]] = load i8, ptr [[GEP037]], align 1 +; CHECK-NEXT: [[I14:%.*]] = add i64 [[I]], 7 +; CHECK-NEXT: [[GEP039:%.*]] = getelementptr inbounds i8, ptr [[PIX1]], i64 [[I14]] +; CHECK-NEXT: [[LOAD040:%.*]] = load i8, ptr [[GEP039]], align 1 +; CHECK-NEXT: [[I15:%.*]] = add i64 [[I1]], 7 +; CHECK-NEXT: [[GEP041:%.*]] = getelementptr inbounds i8, ptr [[PIX2]], i64 [[I15]] +; CHECK-NEXT: [[LOAD042:%.*]] = load i8, ptr [[GEP041]], align 1 +; CHECK-NEXT: [[I16:%.*]] = zext i8 [[LOAD036]] to i32 +; CHECK-NEXT: [[I17:%.*]] = zext i8 [[LOAD028]] to i32 +; CHECK-NEXT: [[I18:%.*]] = zext i8 [[LOAD020]] to i32 +; CHECK-NEXT: [[I19:%.*]] = zext i8 [[LOAD00]] to i32 +; CHECK-NEXT: [[I20:%.*]] = zext i8 [[LOAD042]] to i32 +; CHECK-NEXT: [[I21:%.*]] = zext i8 [[LOAD034]] to i32 +; CHECK-NEXT: [[I22:%.*]] = zext i8 [[LOAD026]] to i32 +; CHECK-NEXT: [[I23:%.*]] = zext i8 [[LOAD018]] to i32 +; CHECK-NEXT: [[I24:%.*]] = zext i8 [[LOAD038]] to i32 +; CHECK-NEXT: [[I25:%.*]] = zext i8 [[LOAD030]] to i32 +; CHECK-NEXT: [[I26:%.*]] = zext i8 [[LOAD022]] to i32 +; CHECK-NEXT: [[I27:%.*]] = zext i8 [[LOAD014]] to i32 +; CHECK-NEXT: [[I28:%.*]] = zext i8 [[LOAD040]] to i32 +; CHECK-NEXT: [[I29:%.*]] = zext i8 [[LOAD032]] to i32 +; CHECK-NEXT: [[I30:%.*]] = zext i8 [[LOAD024]] to i32 +; CHECK-NEXT: [[I31:%.*]] = zext i8 [[LOAD016]] to i32 +; CHECK-NEXT: [[I32:%.*]] = shl i32 [[I20]], 16 +; CHECK-NEXT: [[I33:%.*]] = shl i32 [[I21]], 16 +; CHECK-NEXT: [[I34:%.*]] = shl i32 [[I22]], 16 +; CHECK-NEXT: [[I35:%.*]] = shl i32 [[I23]], 16 +; CHECK-NEXT: [[I36:%.*]] = shl i32 [[I28]], 16 +; CHECK-NEXT: [[I37:%.*]] = shl i32 [[I29]], 16 +; CHECK-NEXT: [[I38:%.*]] = shl i32 [[I30]], 16 +; CHECK-NEXT: [[I39:%.*]] = shl i32 [[I31]], 16 +; CHECK-NEXT: [[N1:%.*]] = sub i32 [[I36]], [[I24]] +; CHECK-NEXT: [[N2:%.*]] = sub i32 [[N1]], [[I32]] +; CHECK-NEXT: [[N3:%.*]] = add i32 [[N2]], [[I16]] +; CHECK-NEXT: [[N4:%.*]] = sub i32 [[I37]], [[I25]] +; CHECK-NEXT: [[N5:%.*]] = sub i32 [[N4]], [[I33]] +; CHECK-NEXT: [[N6:%.*]] = add i32 [[N5]], [[I17]] +; CHECK-NEXT: [[N7:%.*]] = sub i32 [[I38]], [[I26]] +; CHECK-NEXT: [[N8:%.*]] = sub i32 [[N7]], [[I34]] +; CHECK-NEXT: [[N9:%.*]] = add i32 [[N8]], [[I18]] +; CHECK-NEXT: [[N10:%.*]] = sub i32 [[I39]], [[I27]] +; CHECK-NEXT: [[N11:%.*]] = sub i32 [[N10]], [[I35]] +; CHECK-NEXT: [[N12:%.*]] = add i32 [[N11]], [[I19]] ; CHECK-NEXT: [[GEP043:%.*]] = getelementptr inbounds [16 x [8 x i32]], ptr [[R:%.*]], i64 0, i64 [[I1_I64_0]], i64 0 -; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i8>, ptr [[GEP]], align 1 -; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i8>, ptr [[GEP013]], align 1 -; CHECK-NEXT: [[TMP2:%.*]] = load <4 x i8>, ptr [[GEP015]], align 1 -; CHECK-NEXT: [[TMP3:%.*]] = load <4 x i8>, ptr [[GEP017]], align 1 -; CHECK-NEXT: [[TMP4:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i32> -; CHECK-NEXT: [[TMP5:%.*]] = zext <4 x i8> [[TMP3]] to <4 x i32> -; CHECK-NEXT: [[TMP6:%.*]] = zext <4 x i8> [[TMP1]] to <4 x i32> -; CHECK-NEXT: [[TMP7:%.*]] = zext <4 x i8> [[TMP2]] to <4 x i32> -; CHECK-NEXT: [[TMP8:%.*]] = shl <4 x i32> [[TMP5]], splat (i32 16) -; CHECK-NEXT: [[TMP9:%.*]] = shl <4 x i32> [[TMP7]], splat (i32 16) -; CHECK-NEXT: [[TMP10:%.*]] = sub <4 x i32> [[TMP9]], [[TMP6]] -; CHECK-NEXT: [[TMP11:%.*]] = sub <4 x i32> [[TMP10]], [[TMP8]] -; CHECK-NEXT: [[TMP12:%.*]] = add <4 x i32> [[TMP11]], [[TMP4]] -; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <4 x i32> [[TMP12]], <4 x i32> poison, <4 x i32> zeroinitializer -; CHECK-NEXT: [[TMP14:%.*]] = shufflevector <4 x i32> [[TMP12]], <4 x i32> poison, <4 x i32> <i32 1, i32 1, i32 1, i32 1> -; CHECK-NEXT: [[TMP15:%.*]] = add <4 x i32> [[TMP13]], [[TMP14]] -; CHECK-NEXT: [[TMP16:%.*]] = sub <4 x i32> [[TMP13]], [[TMP14]] -; CHECK-NEXT: [[TMP17:%.*]] = shufflevector <4 x i32> [[TMP15]], <4 x i32> [[TMP16]], <4 x i32> <i32 0, i32 5, i32 2, i32 7> -; CHECK-NEXT: [[TMP18:%.*]] = shufflevector <4 x i32> [[TMP12]], <4 x i32> poison, <4 x i32> <i32 2, i32 2, i32 2, i32 2> -; CHECK-NEXT: [[TMP19:%.*]] = add <4 x i32> [[TMP17]], [[TMP18]] -; CHECK-NEXT: [[TMP20:%.*]] = sub <4 x i32> [[TMP17]], [[TMP18]] -; CHECK-NEXT: [[TMP21:%.*]] = shufflevector <4 x i32> [[TMP19]], <4 x i32> [[TMP20]], <4 x i32> <i32 0, i32 1, i32 6, i32 7> -; CHECK-NEXT: [[TMP22:%.*]] = shufflevector <4 x i32> [[TMP12]], <4 x i32> poison, <4 x i32> <i32 3, i32 3, i32 3, i32 3> -; CHECK-NEXT: [[TMP23:%.*]] = add <4 x i32> [[TMP21]], [[TMP22]] -; CHECK-NEXT: [[TMP24:%.*]] = sub <4 x i32> [[TMP21]], [[TMP22]] -; CHECK-NEXT: [[TMP25:%.*]] = shufflevector <4 x i32> [[TMP23]], <4 x i32> [[TMP24]], <4 x i32> <i32 0, i32 5, i32 6, i32 3> -; CHECK-NEXT: store <4 x i32> [[TMP25]], ptr [[GEP043]], align 4 +; CHECK-NEXT: [[N151:%.*]] = add i32 [[N12]], [[N9]] +; CHECK-NEXT: [[N159:%.*]] = add i32 [[N151]], [[N6]] +; CHECK-NEXT: [[N167:%.*]] = add i32 [[N159]], [[N3]] +; CHECK-NEXT: store i32 [[N167]], ptr [[GEP043]], align 4 +; CHECK-NEXT: [[GEP045:%.*]] = getelementptr inbounds [16 x [8 x i32]], ptr [[R]], i64 0, i64 [[I1_I64_0]], i64 2 +; CHECK-NEXT: [[N156:%.*]] = add i32 [[N12]], [[N9]] +; CHECK-NEXT: [[N164:%.*]] = sub i32 [[N156]], [[N6]] +; CHECK-NEXT: [[N172:%.*]] = sub i32 [[N164]], [[N3]] +; CHECK-NEXT: store i32 [[N172]], ptr [[GEP045]], align 4 +; CHECK-NEXT: [[GEP063:%.*]] = getelementptr inbounds [16 x [8 x i32]], ptr [[R]], i64 0, i64 [[I1_I64_0]], i64 1 +; CHECK-NEXT: [[N157:%.*]] = sub i32 [[N12]], [[N9]] +; CHECK-NEXT: [[N165:%.*]] = add i32 [[N157]], [[N6]] +; CHECK-NEXT: [[N173:%.*]] = sub i32 [[N165]], [[N3]] +; CHECK-NEXT: store i32 [[N173]], ptr [[GEP063]], align 4 +; CHECK-NEXT: [[GEP081:%.*]] = getelementptr inbounds [16 x [8 x i32]], ptr [[R]], i64 0, i64 [[I1_I64_0]], i64 3 +; CHECK-NEXT: [[N158:%.*]] = sub i32 [[N12]], [[N9]] +; CHECK-NEXT: [[N166:%.*]] = sub i32 [[N158]], [[N6]] +; CHECK-NEXT: [[N174:%.*]] = add i32 [[N166]], [[N3]] +; CHECK-NEXT: store i32 [[N174]], ptr [[GEP081]], align 4 ; CHECK-NEXT: [[NEXTIVLOOP_1247]] = add nuw nsw i64 [[I1_I64_0]], 1 ; CHECK-NEXT: [[CONDLOOP_1247:%.*]] = icmp ult i64 [[NEXTIVLOOP_1247]], 16 ; CHECK-NEXT: br i1 [[CONDLOOP_1247]], label [[LOOP_1247]], label [[AFTERLOOP_1247:%.*]] @@ -183,16 +251,23 @@ define void @splat_with_gathered_loads(ptr %p, ptr %q, ptr %r) { ; CHECK-LABEL: @splat_with_gathered_loads( ; CHECK-NEXT: entry: ; CHECK-NEXT: [[G0:%.*]] = getelementptr i32, ptr [[P:%.*]], i64 0 +; CHECK-NEXT: [[G1:%.*]] = getelementptr i32, ptr [[P]], i64 3 +; CHECK-NEXT: [[L0:%.*]] = load i32, ptr [[G0]], align 4 +; CHECK-NEXT: [[L1:%.*]] = load i32, ptr [[G1]], align 4 +; CHECK-NEXT: [[X:%.*]] = add i32 [[L0]], 1 +; CHECK-NEXT: [[Y:%.*]] = add i32 [[L1]], 2 +; CHECK-NEXT: [[R0:%.*]] = add i32 [[X]], [[Y]] +; CHECK-NEXT: [[R1:%.*]] = sub i32 [[X]], [[Y]] +; CHECK-NEXT: [[R2:%.*]] = add i32 [[X]], [[Y]] +; CHECK-NEXT: [[R3:%.*]] = sub i32 [[X]], [[Y]] ; CHECK-NEXT: [[S0:%.*]] = getelementptr i32, ptr [[R:%.*]], i64 0 -; CHECK-NEXT: [[TMP0:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 4 [[G0]], <4 x i1> <i1 true, i1 false, i1 false, i1 true>, <4 x i32> poison) -; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP0]], <4 x i32> poison, <2 x i32> <i32 0, i32 3> -; CHECK-NEXT: [[TMP2:%.*]] = add <2 x i32> [[TMP1]], <i32 1, i32 2> -; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <2 x i32> [[TMP2]], <2 x i32> poison, <4 x i32> zeroinitializer -; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <2 x i32> [[TMP2]], <2 x i32> poison, <4 x i32> <i32 1, i32 1, i32 1, i32 1> -; CHECK-NEXT: [[TMP5:%.*]] = add <4 x i32> [[TMP3]], [[TMP4]] -; CHECK-NEXT: [[TMP6:%.*]] = sub <4 x i32> [[TMP3]], [[TMP4]] -; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <4 x i32> [[TMP5]], <4 x i32> [[TMP6]], <4 x i32> <i32 0, i32 5, i32 2, i32 7> -; CHECK-NEXT: store <4 x i32> [[TMP7]], ptr [[S0]], align 4 +; CHECK-NEXT: [[S1:%.*]] = getelementptr i32, ptr [[R]], i64 1 +; CHECK-NEXT: [[S2:%.*]] = getelementptr i32, ptr [[R]], i64 2 +; CHECK-NEXT: [[S3:%.*]] = getelementptr i32, ptr [[R]], i64 3 +; CHECK-NEXT: store i32 [[R0]], ptr [[S0]], align 4 +; CHECK-NEXT: store i32 [[R1]], ptr [[S1]], align 4 +; CHECK-NEXT: store i32 [[R2]], ptr [[S2]], align 4 +; CHECK-NEXT: store i32 [[R3]], ptr [[S3]], align 4 ; CHECK-NEXT: ret void ; entry: @@ -222,15 +297,22 @@ entry: define void @copyable_splat_scalars(ptr %p, ptr %r) { ; CHECK-LABEL: @copyable_splat_scalars( ; CHECK-NEXT: entry: -; CHECK-NEXT: [[TMP0:%.*]] = load <2 x i32>, ptr [[P:%.*]], align 4 -; CHECK-NEXT: [[TMP1:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> poison, <2 x i32> zeroinitializer -; CHECK-NEXT: [[TMP2:%.*]] = add <2 x i32> [[TMP0]], [[TMP1]] -; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <2 x i32> [[TMP2]], <2 x i32> poison, <4 x i32> <i32 1, i32 1, i32 1, i32 1> -; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <2 x i32> [[TMP2]], <2 x i32> poison, <4 x i32> zeroinitializer -; CHECK-NEXT: [[TMP5:%.*]] = add <4 x i32> [[TMP3]], [[TMP4]] -; CHECK-NEXT: [[TMP6:%.*]] = sub <4 x i32> [[TMP3]], [[TMP4]] -; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <4 x i32> [[TMP5]], <4 x i32> [[TMP6]], <4 x i32> <i32 0, i32 5, i32 2, i32 7> -; CHECK-NEXT: store <4 x i32> [[TMP7]], ptr [[R:%.*]], align 4 +; CHECK-NEXT: [[A:%.*]] = load i32, ptr [[P:%.*]], align 4 +; CHECK-NEXT: [[G1:%.*]] = getelementptr i32, ptr [[P]], i64 1 +; CHECK-NEXT: [[B:%.*]] = load i32, ptr [[G1]], align 4 +; CHECK-NEXT: [[X:%.*]] = add i32 [[A]], [[B]] +; CHECK-NEXT: [[Y:%.*]] = shl i32 [[A]], 1 +; CHECK-NEXT: [[R0:%.*]] = add i32 [[X]], [[Y]] +; CHECK-NEXT: [[R1:%.*]] = sub i32 [[X]], [[Y]] +; CHECK-NEXT: [[R2:%.*]] = add i32 [[X]], [[Y]] +; CHECK-NEXT: [[R3:%.*]] = sub i32 [[X]], [[Y]] +; CHECK-NEXT: [[S1:%.*]] = getelementptr i32, ptr [[R:%.*]], i64 1 +; CHECK-NEXT: [[S2:%.*]] = getelementptr i32, ptr [[R]], i64 2 +; CHECK-NEXT: [[S3:%.*]] = getelementptr i32, ptr [[R]], i64 3 +; CHECK-NEXT: store i32 [[R0]], ptr [[R]], align 4 +; CHECK-NEXT: store i32 [[R1]], ptr [[S1]], align 4 +; CHECK-NEXT: store i32 [[R2]], ptr [[S2]], align 4 +; CHECK-NEXT: store i32 [[R3]], ptr [[S3]], align 4 ; CHECK-NEXT: ret void ; entry: @@ -259,14 +341,22 @@ entry: define void @splat_cast_scalars(ptr %p, ptr %r) { ; CHECK-LABEL: @splat_cast_scalars( ; CHECK-NEXT: entry: -; CHECK-NEXT: [[TMP0:%.*]] = load <2 x i16>, ptr [[P:%.*]], align 2 -; CHECK-NEXT: [[TMP1:%.*]] = zext <2 x i16> [[TMP0]] to <2 x i32> -; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <4 x i32> zeroinitializer -; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <4 x i32> <i32 1, i32 1, i32 1, i32 1> -; CHECK-NEXT: [[TMP4:%.*]] = add <4 x i32> [[TMP2]], [[TMP3]] -; CHECK-NEXT: [[TMP5:%.*]] = sub <4 x i32> [[TMP2]], [[TMP3]] -; CHECK-NEXT: [[TMP6:%.*]] = shufflevector <4 x i32> [[TMP4]], <4 x i32> [[TMP5]], <4 x i32> <i32 0, i32 5, i32 2, i32 7> -; CHECK-NEXT: store <4 x i32> [[TMP6]], ptr [[R:%.*]], align 4 +; CHECK-NEXT: [[A:%.*]] = load i16, ptr [[P:%.*]], align 2 +; CHECK-NEXT: [[G1:%.*]] = getelementptr i16, ptr [[P]], i64 1 +; CHECK-NEXT: [[B:%.*]] = load i16, ptr [[G1]], align 2 +; CHECK-NEXT: [[X:%.*]] = zext i16 [[A]] to i32 +; CHECK-NEXT: [[Y:%.*]] = zext i16 [[B]] to i32 +; CHECK-NEXT: [[R0:%.*]] = add i32 [[X]], [[Y]] +; CHECK-NEXT: [[R1:%.*]] = sub i32 [[X]], [[Y]] +; CHECK-NEXT: [[R2:%.*]] = add i32 [[X]], [[Y]] +; CHECK-NEXT: [[R3:%.*]] = sub i32 [[X]], [[Y]] +; CHECK-NEXT: store i32 [[R0]], ptr [[R:%.*]], align 4 +; CHECK-NEXT: [[S1:%.*]] = getelementptr i32, ptr [[R]], i64 1 +; CHECK-NEXT: store i32 [[R1]], ptr [[S1]], align 4 +; CHECK-NEXT: [[S2:%.*]] = getelementptr i32, ptr [[R]], i64 2 +; CHECK-NEXT: store i32 [[R2]], ptr [[S2]], align 4 +; CHECK-NEXT: [[S3:%.*]] = getelementptr i32, ptr [[R]], i64 3 +; CHECK-NEXT: store i32 [[R3]], ptr [[S3]], align 4 ; CHECK-NEXT: ret void ; entry: @@ -294,15 +384,24 @@ entry: define void @splat_cmp_scalars(ptr %p, ptr %r) { ; CHECK-LABEL: @splat_cmp_scalars( ; CHECK-NEXT: entry: -; CHECK-NEXT: [[TMP0:%.*]] = load <2 x i32>, ptr [[P:%.*]], align 4 -; CHECK-NEXT: [[TMP1:%.*]] = icmp ugt <2 x i32> [[TMP0]], <i32 10, i32 20> -; CHECK-NEXT: [[TMP2:%.*]] = zext <2 x i1> [[TMP1]] to <2 x i32> -; CHECK-NEXT: [[TMP3:%.*]] = shufflevector <2 x i32> [[TMP2]], <2 x i32> poison, <4 x i32> zeroinitializer -; CHECK-NEXT: [[TMP4:%.*]] = shufflevector <2 x i32> [[TMP2]], <2 x i32> poison, <4 x i32> <i32 1, i32 1, i32 1, i32 1> -; CHECK-NEXT: [[TMP5:%.*]] = add <4 x i32> [[TMP3]], [[TMP4]] -; CHECK-NEXT: [[TMP6:%.*]] = sub <4 x i32> [[TMP3]], [[TMP4]] -; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <4 x i32> [[TMP5]], <4 x i32> [[TMP6]], <4 x i32> <i32 0, i32 5, i32 2, i32 7> -; CHECK-NEXT: store <4 x i32> [[TMP7]], ptr [[R:%.*]], align 4 +; CHECK-NEXT: [[A:%.*]] = load i32, ptr [[P:%.*]], align 4 +; CHECK-NEXT: [[G1:%.*]] = getelementptr i32, ptr [[P]], i64 1 +; CHECK-NEXT: [[B:%.*]] = load i32, ptr [[G1]], align 4 +; CHECK-NEXT: [[X:%.*]] = icmp ugt i32 [[A]], 10 +; CHECK-NEXT: [[Y:%.*]] = icmp ugt i32 [[B]], 20 +; CHECK-NEXT: [[XA:%.*]] = zext i1 [[X]] to i32 +; CHECK-NEXT: [[YA:%.*]] = zext i1 [[Y]] to i32 +; CHECK-NEXT: [[R0:%.*]] = add i32 [[XA]], [[YA]] +; CHECK-NEXT: [[R1:%.*]] = sub i32 [[XA]], [[YA]] +; CHECK-NEXT: [[R2:%.*]] = add i32 [[XA]], [[YA]] +; CHECK-NEXT: [[R3:%.*]] = sub i32 [[XA]], [[YA]] +; CHECK-NEXT: store i32 [[R0]], ptr [[R:%.*]], align 4 +; CHECK-NEXT: [[S1:%.*]] = getelementptr i32, ptr [[R]], i64 1 +; CHECK-NEXT: store i32 [[R1]], ptr [[S1]], align 4 +; CHECK-NEXT: [[S2:%.*]] = getelementptr i32, ptr [[R]], i64 2 +; CHECK-NEXT: store i32 [[R2]], ptr [[S2]], align 4 +; CHECK-NEXT: [[S3:%.*]] = getelementptr i32, ptr [[R]], i64 3 +; CHECK-NEXT: store i32 [[R3]], ptr [[S3]], align 4 ; CHECK-NEXT: ret void ; entry: _______________________________________________ llvm-branch-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/llvm-branch-commits
