53 "should not try to widen irregular types");
68 auto IsConsecutiveAccess = [&](
VPValue *Addr,
Type *AccessTy) {
77 if (!VPBB->getParent())
80 auto EndIter = Term ? Term->getIterator() : VPBB->end();
85 VPValue *VPV = Ingredient.getVPSingleValue();
106 IsConsecutiveAccess(VPI->getOperand(0), VPI->getScalarType());
108 nullptr , IsConsecutive,
109 *VPI, Ingredient.getDebugLoc());
111 bool IsConsecutive = IsConsecutiveAccess(
112 VPI->getOperand(1), VPI->getOperand(0)->getScalarType());
114 *
Store, Ingredient.getOperand(1), Ingredient.getOperand(0),
115 nullptr , IsConsecutive, *VPI, Ingredient.getDebugLoc());
118 Ingredient.operands(), *VPI,
119 Ingredient.getDebugLoc(),
GEP);
131 if (VectorID == Intrinsic::experimental_noalias_scope_decl)
136 if (VectorID == Intrinsic::assume ||
137 VectorID == Intrinsic::lifetime_end ||
138 VectorID == Intrinsic::lifetime_start ||
139 VectorID == Intrinsic::sideeffect ||
140 VectorID == Intrinsic::pseudoprobe) {
145 const bool IsSingleScalar = VectorID != Intrinsic::assume &&
146 VectorID != Intrinsic::pseudoprobe;
150 Ingredient.getDebugLoc());
153 *CI, VectorID,
drop_end(Ingredient.operands()), CI->getType(),
154 VPIRFlags(*CI), *VPI, CI->getDebugLoc());
158 CI->getOpcode(), Ingredient.getOperand(0), CI->getType(), CI,
162 *VPI, Ingredient.getDebugLoc());
166 "inductions must be created earlier");
175 "Only recpies with zero or one defined values expected");
176 Ingredient.eraseFromParent();
187 const Loop *L =
nullptr;
192 if (
A->getOpcode() != Instruction::Store ||
193 B->getOpcode() != Instruction::Store)
206 const APInt *Distance;
212 Type *TyA =
A->getOperand(0)->getScalarType();
213 uint64_t SizeA =
DL.getTypeStoreSize(TyA);
214 Type *TyB =
B->getOperand(0)->getScalarType();
215 uint64_t SizeB =
DL.getTypeStoreSize(TyB);
220 uint64_t MaxStoreSize = std::max(SizeA, SizeB);
222 auto VFs =
B->getParent()->getPlan()->vectorFactors();
226 return Distance->
abs().
uge(
234 : ExcludeRecipes(ExcludeRecipes.begin(), ExcludeRecipes.end()),
235 GroupLeader(GroupLeader), PSE(&PSE), L(&L) {}
244 return ExcludeRecipes.contains(
Store) ||
245 (
Store && isNoAliasViaDistance(
Store, &GroupLeader));
258 std::optional<SinkStoreInfo> SinkInfo = {}) {
259 bool CheckReads = SinkInfo.has_value();
263 if (SinkInfo && SinkInfo->shouldSkip(R))
267 if (!
R.mayWriteToMemory() && !(CheckReads &&
R.mayReadFromMemory()))
292template <
unsigned Opcode>
297 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
298 "Only Load and Store opcodes supported");
299 constexpr bool IsLoad = (Opcode == Instruction::Load);
302 RecipesByAddressAndType;
307 if (!RepR || RepR->getOpcode() != Opcode || !FilterFn(RepR))
311 VPValue *Addr = RepR->getOperand(IsLoad ? 0 : 1);
315 RecipesByAddressAndType[{AddrSCEV, LoadStoreTy}].push_back(RepR);
320 for (
auto &Group :
Groups) {
335 auto InsertIfValidSinkCandidate = [ScalarVFOnly, &WorkList](
342 if (Candidate->getParent() == SinkTo ||
343 all_of(Candidate->operands(),
344 [](
VPValue *
Op) { return Op->isDefinedOutsideLoopRegions(); }) ||
356 WorkList.
insert({SinkTo, Candidate});
368 for (
auto &Recipe : *VPBB)
370 InsertIfValidSinkCandidate(VPBB,
Op);
374 for (
unsigned I = 0;
I != WorkList.
size(); ++
I) {
377 std::tie(SinkTo, SinkCandidate) = WorkList[
I];
382 auto UsersOutsideSinkTo =
384 return cast<VPRecipeBase>(U)->getParent() != SinkTo;
386 if (
any_of(UsersOutsideSinkTo, [SinkCandidate](
VPUser *U) {
387 return !U->usesFirstLaneOnly(SinkCandidate);
390 bool NeedsDuplicating = !UsersOutsideSinkTo.empty();
392 if (NeedsDuplicating) {
396 if (
auto *SinkCandidateRepR =
401 SinkCandidateRepR->getOpcode(), SinkCandidate->
operands(),
402 nullptr, *SinkCandidateRepR, *SinkCandidateRepR,
406 Clone = SinkCandidate->
clone();
416 InsertIfValidSinkCandidate(SinkTo,
Op);
425 if (EntryBB->getNumSuccessors() != 2)
430 if (!Succ0 || !Succ1)
433 if (Succ0->getNumSuccessors() + Succ1->getNumSuccessors() != 1)
435 if (Succ0->getSingleSuccessor() == Succ1)
437 if (Succ1->getSingleSuccessor() == Succ0)
454 if (!Region1->isReplicator())
456 auto *MiddleBasicBlock =
458 if (!MiddleBasicBlock || !MiddleBasicBlock->empty())
463 if (!Region2 || !Region2->isReplicator())
466 VPValue *Mask1 = Region1->getEntryBranchOnMask()->getOperand(0);
467 VPValue *Mask2 = Region2->getEntryBranchOnMask()->getOperand(0);
468 if (!Mask1 || Mask1 != Mask2)
471 assert(Mask1 && Mask2 &&
"both region must have conditions");
477 if (TransformedRegions.
contains(Region1))
484 if (!Then1 || !Then2)
504 VPValue *Phi1ToMoveV = Phi1ToMove.getVPSingleValue();
510 if (Phi1ToMove.getVPSingleValue()->user_empty()) {
511 Phi1ToMove.eraseFromParent();
514 Phi1ToMove.moveBefore(*Merge2, Merge2->begin());
528 TransformedRegions.
insert(Region1);
531 return !TransformedRegions.
empty();
539 std::string RegionName = (
Twine(
"pred.") + Instr->getOpcodeName()).str();
540 assert(Instr->getParent() &&
"Predicated instruction not in any basic block");
541 auto *BlockInMask = PredRecipe->
getMask();
562 Region->setParent(ParentRegion);
568 RecipeWithoutMask->getDebugLoc());
569 Exiting->appendRecipe(PHIRecipe);
582 if (RepR->isPredicated())
601 if (ParentRegion && ParentRegion->
getExiting() == CurrentBlock)
613 if (!VPBB->getParent())
617 if (!PredVPBB || PredVPBB->getNumSuccessors() != 1 ||
626 R.moveBefore(*PredVPBB, PredVPBB->
end());
628 auto *ParentRegion = VPBB->getParent();
629 if (ParentRegion && ParentRegion->getExiting() == VPBB)
630 ParentRegion->setExiting(PredVPBB);
634 return !WorkList.
empty();
641 bool ShouldSimplify =
true;
642 while (ShouldSimplify) {
658 if (!
IV ||
IV->getTruncInst())
673 for (
auto *U : FindMyCast->
users()) {
675 if (UserCast && UserCast->getUnderlyingValue() == IRCast) {
676 FoundUserCast = UserCast;
683 FindMyCast = FoundUserCast;
685 if (FindMyCast !=
IV)
709 PhiR->replaceAllUsesWith(PhiR->getOperand(0));
711 PhiR->eraseFromParent();
767 Def->user_empty() || !Def->getUnderlyingValue() ||
768 (RepR && (RepR->isSingleScalar() || RepR->isPredicated())))
781 Def->getUnderlyingInstr()->getOpcode(), Def->operands(),
783 Def->getUnderlyingInstr());
784 Clone->insertAfter(Def);
785 Def->replaceAllUsesWith(Clone);
786 Def->eraseFromParent();
801 PtrIV->replaceAllUsesWith(PtrAdd);
808 if (HasOnlyVectorVFs &&
none_of(WideIV->users(), [WideIV](
VPUser *U) {
809 return U->usesScalars(WideIV);
818 WrapFlags = {
static_cast<bool>(WideIV->getNoWrapFlagsOrNone().HasNUW),
821 Plan, ID.getKind(), ID.getInductionOpcode(),
823 WideIV->getTruncInst(), WideIV->getStartValue(), WideIV->getStepValue(),
824 WideIV->getDebugLoc(), Builder, WrapFlags);
827 if (!HasOnlyVectorVFs) {
829 "plans containing a scalar VF cannot also include scalable VFs");
830 WideIV->replaceAllUsesWith(Steps);
833 WideIV->replaceUsesWithIf(Steps,
834 [WideIV, HasScalableVF](
VPUser &U,
unsigned) {
836 return U.usesFirstLaneOnly(WideIV);
837 return U.usesScalars(WideIV);
853 return (IntOrFpIV && IntOrFpIV->getTruncInst()) ? nullptr : WideIV;
858 if (!Def || Def->getNumOperands() != 2)
866 auto IsWideIVInc = [&]() {
867 auto &ID = WideIV->getInductionDescriptor();
870 VPValue *IVStep = WideIV->getStepValue();
871 switch (ID.getInductionOpcode()) {
872 case Instruction::Add:
874 case Instruction::FAdd:
876 case Instruction::FSub:
879 case Instruction::Sub: {
899 return IsWideIVInc() ? WideIV :
nullptr;
923 VPValue *FirstActiveLane =
B.createFirstActiveLane(Mask,
DL);
925 B.createScalarZExtOrTrunc(FirstActiveLane, CanonicalIVType,
DL);
926 VPValue *EndValue =
B.createAdd(CanonicalIV, FirstActiveLane,
DL);
931 if (Incoming != WideIV) {
933 EndValue =
B.createAdd(EndValue, One,
DL);
938 VPIRValue *Start = WideIV->getStartValue();
939 VPValue *Step = WideIV->getStepValue();
940 EndValue =
B.createDerivedIV(
942 Start, EndValue, Step);
956 if (WideIntOrFp && WideIntOrFp->getTruncInst())
966 Start, VectorTC, Step);
998 assert(EndValue &&
"Must have computed the end value up front");
1003 if (Incoming != WideIV)
1015 auto *Zero = Plan.
getZero(StepTy);
1016 return B.createPtrAdd(EndValue,
B.createSub(Zero, Step),
1021 return B.createNaryOp(
1022 ID.getInductionBinOp()->getOpcode() == Instruction::FAdd
1024 : Instruction::FAdd,
1025 {EndValue, Step}, {ID.getInductionBinOp()->getFastMathFlags()});
1040 const SCEV *Start, *Step;
1058 VPValue *ExitCount = Builder.createOverflowingOp(
1061 return Builder.createDerivedIV(Kind,
nullptr, StartVPV, ExitCount,
1070 VPBuilder VectorPHBuilder(VectorPH, VectorPH->getFirstNonPhi());
1080 EndValues[WideIV] = EndValue;
1090 R.getVPSingleValue()->replaceAllUsesWith(EndValue);
1091 R.eraseFromParent();
1100 for (
auto [Idx, PredVPBB] :
enumerate(ExitVPBB->getPredecessors())) {
1102 if (PredVPBB == MiddleVPBB) {
1104 Plan, ExitIRI->getOperand(Idx), EndValues, PSE);
1107 Plan, ExitIRI->getOperand(Idx), PSE, ResumeTC, L);
1110 Plan, ExitIRI->getOperand(Idx), PSE);
1113 ExitIRI->setOperand(Idx, Escape);
1130 const auto &[V, Inserted] = SCEV2VPV.
try_emplace(ExpR->getSCEV(), ExpR);
1134 ExpR->replaceAllUsesWith(V->second);
1138 ExpR->eraseFromParent();
1145 bool CanCreateNewRecipe) {
1146 VPlan *Plan = Def->getParent()->getPlan();
1172 return Plan->
getZero(Def->getScalarType());
1187 if (CanCreateNewRecipe &&
1192 (!Def->getOperand(0)->hasMoreThanOneUniqueUser() ||
1193 !Def->getOperand(1)->hasMoreThanOneUniqueUser()))
1194 return Builder.createLogicalAnd(
X, Builder.createOr(
Y, Z));
1199 return Def->getOperand(1);
1204 return Builder.createLogicalAnd(
X,
Y);
1215 if (CanCreateNewRecipe &&
1217 return Builder.createNot(
C);
1221 Def->setOperand(0,
C);
1222 Def->setOperand(1,
Y);
1223 Def->setOperand(2,
X);
1228 if (CanCreateNewRecipe &&
1232 Y->getScalarType()->isIntegerTy(1))
1233 return Builder.createOr(
Y, Builder.createLogicalAnd(
X, Z));
1237 if (CanCreateNewRecipe &&
1243 return Builder.createSelect(Builder.createLogicalAnd(Mask0, Mask1),
X,
Y,
1244 Def->getDebugLoc());
1253 VPlan *Plan = Def->getParent()->getPlan();
1273 RepR && RepR->isPredicated() && RepR->getOpcode() == Instruction::Store &&
1277 RepR->getUnderlyingInstr(), RepR->operandsWithoutMask(),
1278 RepR->isSingleScalar(),
nullptr, *RepR, *RepR,
1279 RepR->getDebugLoc());
1280 Unmasked->insertBefore(RepR);
1294 bool CanCreateNewRecipe =
1301 Def->getScalarType() ==
A->getScalarType())
1305 Type *TruncTy = Def->getScalarType();
1306 Type *ATy =
A->getScalarType();
1307 if (TruncTy == ATy) {
1316 : Instruction::ZExt;
1319 if (
auto *UnderlyingExt = Z->getUnderlyingValue()) {
1321 Ext->setUnderlyingValue(UnderlyingExt);
1325 auto *Trunc = Builder.createWidenCast(Instruction::Trunc,
A, TruncTy);
1342 return Plan->
getZero(Def->getScalarType());
1348 return Builder.createSub(Plan->
getZero(
A->getScalarType()),
A,
1349 Def->getDebugLoc(),
"", NW);
1352 if (CanCreateNewRecipe &&
1360 return Builder.createSub(
X,
Y, Def->getDebugLoc(),
"", NW);
1367 Def->getDebugLoc());
1374 MulR->hasNoSignedWrap() &&
1376 return Builder.createNaryOp(
1379 Def->getDebugLoc());
1384 return Builder.createNaryOp(
1400 return match(U, m_Not(m_Specific(Cmp))) ||
1401 (match(U, m_Select(m_Specific(Cmp), m_VPValue(),
1403 U->getOperand(1) != Cmp && U->getOperand(2) != Cmp);
1410 R->setOperand(1,
Y);
1411 R->setOperand(2,
X);
1415 R->replaceAllUsesWith(Cmp);
1420 if (!Cmp->getDebugLoc() && Def->getDebugLoc())
1421 Cmp->setDebugLoc(Def->getDebugLoc());
1434 if (
Op->getNumUsers() > 1 ||
1438 }
else if (!UnpairedCmp) {
1439 UnpairedCmp =
Op->getDefiningRecipe();
1443 UnpairedCmp =
nullptr;
1450 if (NewOps.
size() < Def->getNumOperands()) {
1459 if (CanCreateNewRecipe &&
1470 A->getScalarType() == Def->getScalarType())
1475 Type *WideStepTy = Def->getScalarType();
1476 if (
X->getScalarType() != WideStepTy)
1477 X = Builder.createWidenCast(Instruction::Trunc,
X, WideStepTy);
1486 Def->getScalarType()->isIntegerTy(1)) {
1487 Def->setOperand(1, Plan->
getTrue());
1488 Def->setOperand(0,
Y);
1495 return Def->getOperand(0);
1501 return BuildVector->getOperand(BuildVector->getNumOperands() - 1);
1517 return BuildVector->getOperand(BuildVector->getNumOperands() - 2);
1523 return BuildVector->getOperand(Idx);
1532 Def->replaceUsesWithIf(Def->getOperand(0), [Def](
VPUser &U,
unsigned) {
1533 return U.usesFirstLaneOnly(Def);
1543 "broadcast operand must be single-scalar");
1544 Def->setOperand(0, Z);
1549 Def->replaceUsesWithIf(
1550 X, [Def](
const VPUser &U,
unsigned) {
return U.usesScalars(Def); });
1555 if (Def->getNumOperands() == 1) {
1556 return Def->getOperand(0);
1560 return Phi->getOperand(0);
1566 if (Def->getNumOperands() == 1 &&
1591 return Builder.createNaryOp(Instruction::ExtractElement, {
A, LaneToExtract},
1592 Def->getDebugLoc());
1606 if (IVInc->getNumUsers() == 2) {
1611 if (Phi->getNumUsers() == 1 || (Phi->getNumUsers() == 2 && Inc)) {
1612 Def->replaceAllUsesWith(IVInc);
1614 Inc->replaceAllUsesWith(Phi);
1615 Phi->setOperand(0,
Y);
1625 return VPR->getOperand(0);
1631 return Steps->getOperand(0);
1637 Def->replaceUsesWithIf(StartV, [](
const VPUser &U,
unsigned Idx) {
1639 return PhiR && PhiR->isInLoop();
1659 Def->replaceAllUsesWith(New);
1660 Def->eraseFromParent();
1663 Def->eraseFromParent();
1682 R.getVPSingleValue()->replaceAllUsesWith(
X);
1698 while (!Worklist.
empty()) {
1707 R->replaceAllUsesWith(
1708 Builder.createLogicalAnd(HeaderMask, Builder.createLogicalAnd(
X,
Y)));
1712static std::optional<Instruction::BinaryOps>
1715 case Intrinsic::masked_udiv:
1716 return Instruction::UDiv;
1717 case Intrinsic::masked_sdiv:
1718 return Instruction::SDiv;
1719 case Intrinsic::masked_urem:
1720 return Instruction::URem;
1721 case Intrinsic::masked_srem:
1722 return Instruction::SRem;
1739 if (RepR && (RepR->isSingleScalar() || RepR->isPredicated()))
1743 if (RepR && RepR->getOpcode() == Instruction::Store &&
1746 RepOrWidenR->getUnderlyingInstr(), RepOrWidenR->operands(),
1747 true ,
nullptr , *RepR ,
1748 *RepR , RepR->getDebugLoc());
1749 Clone->insertBefore(RepOrWidenR);
1751 VPValue *ExtractOp = Clone->getOperand(0);
1757 Clone->setOperand(0, ExtractOp);
1758 RepR->eraseFromParent();
1770 VPValue *SafeDivisor = Builder.createSelect(
1771 IntrR->getOperand(2), IntrR->getOperand(1),
1773 VPValue *Clone = Builder.createNaryOp(
1774 *
Opc, {IntrR->getOperand(0), SafeDivisor},
1777 IntrR->eraseFromParent();
1786 auto IntroducesBCastOf = [](
const VPValue *
Op) {
1795 return !U->usesScalars(
Op);
1799 if (
any_of(RepOrWidenR->users(), IntroducesBCastOf(RepOrWidenR)) &&
1802 make_filter_range(Op->users(), not_equal_to(RepOrWidenR)),
1803 IntroducesBCastOf(Op)))
1807 bool LiveInNeedsBroadcast =
1808 isa<VPIRValue>(Op) && !isa<VPConstant>(Op);
1809 auto *OpR = dyn_cast<VPReplicateRecipe>(Op);
1810 return LiveInNeedsBroadcast || (OpR && OpR->isSingleScalar());
1817 RepOrWidenR->getUnderlyingInstr());
1818 Clone->insertBefore(RepOrWidenR);
1819 RepOrWidenR->replaceAllUsesWith(Clone);
1821 RepOrWidenR->eraseFromParent();
1857 if (Blend->isNormalized() || !
match(Blend->getMask(0),
m_False()))
1858 UniqueValues.
insert(Blend->getIncomingValue(0));
1859 for (
unsigned I = 1;
I != Blend->getNumIncomingValues(); ++
I)
1861 UniqueValues.
insert(Blend->getIncomingValue(
I));
1863 if (UniqueValues.
size() == 1) {
1864 Blend->replaceAllUsesWith(*UniqueValues.
begin());
1865 Blend->eraseFromParent();
1869 if (Blend->isNormalized())
1875 unsigned StartIndex = 0;
1876 for (
unsigned I = 0;
I != Blend->getNumIncomingValues(); ++
I) {
1888 OperandsWithMask.
push_back(Blend->getIncomingValue(StartIndex));
1890 for (
unsigned I = 0;
I != Blend->getNumIncomingValues(); ++
I) {
1891 if (
I == StartIndex)
1893 OperandsWithMask.
push_back(Blend->getIncomingValue(
I));
1894 OperandsWithMask.
push_back(Blend->getMask(
I));
1899 OperandsWithMask, *Blend, Blend->getDebugLoc());
1900 NewBlend->insertBefore(&R);
1902 VPValue *DeadMask = Blend->getMask(StartIndex);
1904 Blend->eraseFromParent();
1909 if (NewBlend->getNumOperands() == 3 &&
1911 VPValue *Inc0 = NewBlend->getOperand(0);
1912 VPValue *Inc1 = NewBlend->getOperand(1);
1913 VPValue *OldMask = NewBlend->getOperand(2);
1914 NewBlend->setOperand(0, Inc1);
1915 NewBlend->setOperand(1, Inc0);
1916 NewBlend->setOperand(2, NewMask);
1943 APInt MaxVal = AlignedTC - 1;
1946 unsigned NewBitWidth =
1952 bool MadeChange =
false;
1977 "canonical IV is not expected to have a truncation");
1982 NewWideIV->insertBefore(WideIV);
1989 Cmp->replaceAllUsesWith(
1990 VPBuilder(Cmp).createICmp(Cmp->getPredicate(), NewWideIV, NewBTC));
2004 return any_of(
Cond->getDefiningRecipe()->operands(), [&Plan, BestVF, BestUF,
2006 return isConditionTrueViaVFAndUF(C, Plan, BestVF, BestUF, PSE);
2020 const SCEV *VectorTripCount =
2025 "Trip count SCEV must be computable");
2040 bool MadeChange =
false;
2048 for (
VPBasicBlock *VPBB : {PreheaderVPBB, ExitingVPBB}) {
2057 Builder.setInsertPoint(Extract);
2060 Start = Builder.createAdd(
2065 Extract->eraseFromParent();
2080 auto *Term = &ExitingVPBB->
back();
2086 bool MatchedCanIVInc =
2092 if (MatchedCanIVInc ||
2100 const SCEV *VectorTripCount =
2106 "Trip count SCEV must be computable");
2125 Term->setOperand(1, Plan.
getTrue());
2130 {}, Term->getDebugLoc());
2132 Term->eraseFromParent();
2140 assert(Plan.
hasVF(BestVF) &&
"BestVF is not available in Plan");
2141 assert(Plan.
hasUF(BestUF) &&
"BestUF is not available in Plan");
2160 RecurKind RK = PhiR->getRecurrenceKind();
2167 RecWithFlags->dropPoisonGeneratingFlags();
2173struct VPCSEDenseMapInfo :
public DenseMapInfo<VPSingleDefRecipe *> {
2182 return GEP->getSourceElementType();
2185 .Case<VPVectorPointerRecipe, VPWidenGEPRecipe>(
2186 [](
auto *
I) {
return I->getSourceElementType(); })
2187 .
Default([](
auto *) {
return nullptr; });
2191 static bool canHandle(
const VPSingleDefRecipe *Def) {
2200 if (!
C || (!
C->first && (
C->second == Instruction::InsertValue ||
2201 C->second == Instruction::ExtractValue)))
2205 return !
Def->mayReadOrWriteMemory();
2209 static unsigned getHashValue(
const VPSingleDefRecipe *Def) {
2212 getGEPSourceElementType(Def),
Def->getScalarType(),
2215 if (RFlags->hasPredicate())
2218 return hash_combine(Result, SIVSteps->getInductionOpcode());
2223 static bool isEqual(
const VPSingleDefRecipe *L,
const VPSingleDefRecipe *R) {
2224 if (
L->getVPRecipeID() !=
R->getVPRecipeID() ||
2227 getGEPSourceElementType(L) != getGEPSourceElementType(R) ||
2229 !
equal(
L->operands(),
R->operands()))
2233 "must have valid opcode info for both recipes");
2235 if (LFlags->hasPredicate() &&
2236 LFlags->getPredicate() !=
2240 if (LSIV->getInductionOpcode() !=
2250 const VPRegionBlock *RegionL =
L->getRegion();
2251 const VPRegionBlock *RegionR =
R->getRegion();
2254 L->getParent() !=
R->getParent())
2256 return L->getScalarType() ==
R->getScalarType();
2272 if (!Def || !VPCSEDenseMapInfo::canHandle(Def))
2276 if (!VPDT.
dominates(V->getParent(), VPBB))
2281 Def->replaceAllUsesWith(V);
2294 bool Sinking =
false) {
2323 "Expected vector prehader's successor to be the vector loop region");
2331 return !Op->isDefinedOutsideLoopRegions();
2334 R.moveBefore(*Preheader, Preheader->
end());
2354 assert(!RepR->isPredicated() &&
2355 "Expected prior transformation of predicated replicates to "
2356 "replicate regions");
2361 if (!RepR->isSingleScalar())
2365 if (RepR->getOpcode() == Instruction::Store &&
2366 !RepR->getOperand(1)->isDefinedOutsideLoopRegions())
2371 assert((!R.mayWriteToMemory() ||
2372 (RepR && RepR->getOpcode() == Instruction::Store &&
2373 RepR->getOperand(1)->isDefinedOutsideLoopRegions())) &&
2374 "The only recipes that may write to memory are expected to be "
2375 "stores with invariant pointer-operand");
2385 if (
any_of(Def->users(), [&SinkBB, &LoopRegion](
VPUser *U) {
2386 auto *UserR = cast<VPRecipeBase>(U);
2387 VPBasicBlock *Parent = UserR->getParent();
2389 if (SinkBB && SinkBB != Parent)
2394 return UserR->isPhi() || Parent->getEnclosingLoopRegion() ||
2395 Parent->getSinglePredecessor() != LoopRegion;
2405 "Defining block must dominate sink block");
2430 VPValue *ResultVPV = R.getVPSingleValue();
2432 unsigned NewResSizeInBits = MinBWs.
lookup(UI);
2433 if (!NewResSizeInBits)
2446 (void)OldResSizeInBits;
2454 VPW->dropPoisonGeneratingFlags();
2456 assert((OldResSizeInBits != NewResSizeInBits ||
2458 "Only ICmps should not need extending the result.");
2464 if (OldResSizeInBits != NewResSizeInBits) {
2466 Instruction::ZExt, ResultVPV, OldResTy);
2468 Ext->setOperand(0, ResultVPV);
2478 unsigned OpSizeInBits =
Op->getScalarType()->getScalarSizeInBits();
2479 if (OpSizeInBits == NewResSizeInBits)
2481 assert(OpSizeInBits > NewResSizeInBits &&
"nothing to truncate");
2482 auto [ProcessedIter, Inserted] = ProcessedTruncs.
try_emplace(
Op);
2488 Builder.setInsertPoint(&R);
2489 ProcessedIter->second =
2490 Builder.createWidenCast(Instruction::Trunc,
Op, NewResTy);
2492 Op = ProcessedIter->second;
2496 NWR->insertBefore(&R);
2500 VPValue *Replacement = NWR->getVPSingleValue();
2501 if (OldResSizeInBits != NewResSizeInBits)
2507 R.eraseFromParent();
2513 std::optional<VPDominatorTree> VPDT;
2521 bool SimplifiedPhi =
false;
2531 assert(VPBB->getNumSuccessors() == 2 &&
2532 "Two successors expected for BranchOnCond");
2533 unsigned RemovedIdx;
2544 "There must be a single edge between VPBB and its successor");
2549 SimplifiedPhi =
true;
2553 if (!PhiR || PhiR->getNumIncoming() != 1)
2555 PhiR->replaceAllUsesWith(PhiR->getOperand(0));
2556 PhiR->eraseFromParent();
2561 VPBB->back().eraseFromParent();
2573 if (Reachable.contains(
B))
2584 for (
VPValue *Def : R.definedValues())
2585 Def->replaceAllUsesWith(&Tmp);
2586 R.eraseFromParent();
2590 return SimplifiedPhi;
2616 auto GetSimplifiedLiveInViaSCEV = [&](
VPValue *VPV) ->
VPValue * {
2625 if (
VPValue *SimplifiedLiveIn = GetSimplifiedLiveInViaSCEV(LiveIn))
2626 LiveIn->replaceAllUsesWith(SimplifiedLiveIn);
2637 "expected to run before loop regions are created");
2639 auto CanUseVersionedStride = [&VPDT, Header = Header, &Plan](
VPUser &U,
2646 return VPDT.
dominates(Header, R->getParent());
2650 Value *StrideV = Stride->getValue();
2651 const APInt *StrideConst;
2658 CanUseVersionedStride);
2672 CanUseVersionedStride);
2674 RewriteMap[StrideV] = StrideExpr;
2681 const SCEV *ScevExpr = ExpSCEV->getSCEV();
2684 if (NewSCEV != ScevExpr) {
2686 ExpSCEV->replaceAllUsesWith(NewExp);
2697 auto CollectPoisonGeneratingInstrsInBackwardSlice([&](
VPRecipeBase *Root) {
2702 while (!Worklist.
empty()) {
2705 if (!Visited.
insert(CurRec).second)
2727 RecWithFlags->isDisjoint()) {
2730 Builder.createAdd(
A,
B, RecWithFlags->getDebugLoc());
2731 New->setUnderlyingValue(RecWithFlags->getUnderlyingValue());
2732 RecWithFlags->replaceAllUsesWith(New);
2733 RecWithFlags->eraseFromParent();
2736 RecWithFlags->dropPoisonGeneratingFlags();
2741 assert((!Instr || !Instr->hasPoisonGeneratingFlags()) &&
2742 "found instruction with poison generating flags not covered by "
2743 "VPRecipeWithIRFlags");
2748 if (
VPRecipeBase *OpDef = Operand->getDefiningRecipe())
2770 VPRecipeBase *AddrDef = WidenRec->getAddr()->getDefiningRecipe();
2771 if (AddrDef && WidenRec->isConsecutive() && WidenRec->getMask() &&
2772 match(WidenRec->getMask(), m_UnlessHdrMask))
2773 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2775 VPRecipeBase *AddrDef = InterleaveRec->getAddr()->getDefiningRecipe();
2776 if (AddrDef && InterleaveRec->getMask() &&
2777 match(InterleaveRec->getMask(), m_UnlessHdrMask))
2778 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2788 const bool &EpilogueAllowed) {
2789 if (InterleaveGroups.empty())
2800 IRMemberToRecipe[&MemR->getIngredient()] = MemR;
2807 for (
const auto *IG : InterleaveGroups) {
2810 for (
auto *Member : IG->members())
2812 StartMember = Member;
2820 for (
unsigned I = 0;
I < IG->getFactor(); ++
I) {
2826 StoredValues.
push_back(StoreR->getStoredValue());
2833 bool NeedsMaskForGaps =
2834 (IG->requiresScalarEpilogue() && !EpilogueAllowed) ||
2835 (!StoredValues.
empty() && !IG->isFull());
2838 auto *InsertPos = IRMemberToRecipe.
lookup(IRInsertPos);
2842 "Dead member in non-load group?");
2847 InsertPos->getAsRecipe()))
2848 InsertPos = MemberR;
2849 IRInsertPos = &InsertPos->getIngredient();
2859 VPValue *Addr = Start->getAddr();
2861 if (IG->getIndex(StartMember) != 0 ||
2869 assert(IG->getIndex(IRInsertPos) != 0 &&
2870 "index of insert position shouldn't be zero");
2874 IG->getIndex(IRInsertPos),
2878 Addr =
B.createNoWrapPtrAdd(InsertPos->getAddr(), OffsetVPV, NW);
2884 if (IG->isReverse()) {
2887 -(int64_t)IG->getFactor(), NW, InsertPosR->
getDebugLoc());
2888 ReversePtr->insertBefore(InsertPosR);
2892 IG, Addr, StoredValues, InsertPos->getMask(), NeedsMaskForGaps,
2894 VPIG->insertBefore(InsertPosR);
2897 for (
unsigned i = 0; i < IG->getFactor(); ++i)
2900 if (!Member->getType()->isVoidTy()) {
2918static std::optional<VPValue *>
2971 VPValue *UncountableCondition =
nullptr;
2975 return std::nullopt;
2978 Worklist.
push_back(UncountableCondition);
2979 while (!Worklist.
empty()) {
2983 if (V->isDefinedOutsideLoopRegions())
2989 if (V->getNumUsers() > 1)
2990 return std::nullopt;
3002 return std::nullopt;
3006 return std::nullopt;
3014 return std::nullopt;
3019 if (Recipes.
empty() ||
3021 return std::nullopt;
3023 return UncountableCondition;
3079 for (
auto &Exit : Exits) {
3080 if (Exit.EarlyExitingVPBB == LatchVPBB)
3084 cast<VPIRPhi>(&R)->removeIncomingValueFor(Exit.EarlyExitingVPBB);
3085 Exit.EarlyExitingVPBB->getTerminator()->eraseFromParent();
3096 std::optional<VPValue *>
Cond =
3112 assert(
Load &&
"Couldn't find exactly one load");
3115 "Uncountable exit condition load is conditional.");
3129 DL.getTypeStoreSize(
Load->getScalarType()).getFixedValue());
3153 while (InsertIt != HeaderVPBB->
end() &&
3155 erase(ConditionRecipes, &*InsertIt);
3158 for (
auto *Recipe :
reverse(ConditionRecipes))
3159 Recipe->moveBefore(*HeaderVPBB, InsertIt);
3163 VPBuilder MaskBuilder(HeaderVPBB, InsertIt);
3165 Type *IVScalarTy =
IV->getScalarType();
3171 "uncountable.exit.mask");
3176 if (R.mayReadOrWriteMemory() && &R !=
Load) {
3178 if (!VPDT.
dominates(R.getParent(), LatchVPBB))
3188 "Expected BranchOnCond terminator for MiddleVPBB");
3199 auto Phis = ScalarPH->
phis();
3209 "Continuing from different IV");
3231 VPBuilder LatchBuilder(LatchVPBB->getTerminator());
3233 for (
auto [EarlyExitingVPBB, ExitBlock] :
3237 VPValue *CondOfEarlyExitingVPBB;
3238 [[maybe_unused]]
bool Matched =
3239 match(EarlyExitingVPBB->getTerminator(),
3241 assert(Matched &&
"Terminator must be BranchOnCond");
3245 VPBuilder EarlyExitingBuilder(EarlyExitingVPBB->getTerminator());
3246 auto *CondToEarlyExit = EarlyExitingBuilder.
createNaryOp(
3248 TrueSucc == ExitBlock
3249 ? CondOfEarlyExitingVPBB
3250 : EarlyExitingBuilder.
createNot(CondOfEarlyExitingVPBB));
3256 "exit condition must dominate the latch");
3264 assert(!Exits.
empty() &&
"must have at least one early exit");
3271 for (
const auto &[Num, VPB] :
enumerate(RPOT))
3274 return RPOIdx[
A.EarlyExitingVPBB] < RPOIdx[
B.EarlyExitingVPBB];
3280 for (
unsigned I = 0;
I + 1 < Exits.
size(); ++
I)
3281 for (
unsigned J =
I + 1; J < Exits.
size(); ++J)
3283 Exits[
I].EarlyExitingVPBB) &&
3284 "RPO sort must place dominating exits before dominated ones");
3290 VPValue *Combined = Exits[0].CondToExit;
3303 "Unexpected terminator");
3304 VPValue *IsLatchExitTaken = LatchExitingBranch->getOperand(0);
3305 DebugLoc LatchDL = LatchExitingBranch->getDebugLoc();
3306 LatchExitingBranch->eraseFromParent();
3309 {IsAnyExitTaken, IsLatchExitTaken}, LatchDL);
3310 LatchVPBB->clearSuccessors();
3315 LatchVPBB->setSuccessors({MiddleVPBB, MiddleVPBB, HeaderVPBB});
3316 MiddleVPBB->clearPredecessors();
3317 MiddleVPBB->setPredecessors({LatchVPBB, LatchVPBB});
3319 Plan, Exits, HeaderVPBB, LatchVPBB, MiddleVPBB, TheLoop, PSE, DT, AC);
3324 for (
unsigned Idx = 0; Idx != Exits.
size(); ++Idx) {
3328 VectorEarlyExitVPBBs[Idx] = VectorEarlyExitVPBB;
3336 Exits.
size() == 1 ? VectorEarlyExitVPBBs[0]
3339 LatchVPBB->setSuccessors({DispatchVPBB, MiddleVPBB, HeaderVPBB});
3371 for (
auto [Exit, VectorEarlyExitVPBB] :
3372 zip_equal(Exits, VectorEarlyExitVPBBs)) {
3373 auto &[EarlyExitingVPBB, EarlyExitVPBB,
_] = Exit;
3385 ExitIRI->getIncomingValueForBlock(EarlyExitingVPBB);
3386 VPValue *NewIncoming = IncomingVal;
3388 VPBuilder EarlyExitBuilder(VectorEarlyExitVPBB);
3393 ExitIRI->removeIncomingValueFor(EarlyExitingVPBB);
3394 ExitIRI->addIncoming(NewIncoming);
3397 EarlyExitingVPBB->getTerminator()->eraseFromParent();
3431 bool IsLastDispatch = (
I + 2 == Exits.
size());
3433 IsLastDispatch ? VectorEarlyExitVPBBs.
back()
3439 VectorEarlyExitVPBBs[
I]->setPredecessors({CurrentBB});
3442 CurrentBB = FalseBB;
3457 VPValue *VecOp = Red->getVecOp();
3459 assert(!Red->isPartialReduction() &&
3460 "This path does not support partial reductions");
3463 auto IsExtendedRedValidAndClampRange =
3476 "getExtendedReductionCost only supports integer types");
3477 ExtRedCost = Ctx.TTI.getExtendedReductionCost(
3478 Opcode, ExtOpc == Instruction::CastOps::ZExt, RedTy, SrcVecTy,
3479 Red->getFastMathFlagsOrNone(),
CostKind);
3480 return ExtRedCost.
isValid() && ExtRedCost < ExtCost + RedCost;
3488 IsExtendedRedValidAndClampRange(
3509 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3510 Opcode != Instruction::FAdd)
3513 assert(!Red->isPartialReduction() &&
3514 "This path does not support partial reductions");
3518 auto IsMulAccValidAndClampRange =
3530 (Ext0->getOpcode() != Ext1->getOpcode() ||
3531 Ext0->getOpcode() == Instruction::CastOps::FPExt))
3535 !Ext0 || Ext0->getOpcode() == Instruction::CastOps::ZExt;
3537 MulAccCost = Ctx.TTI.getMulAccReductionCost(IsZExt, Opcode, RedTy,
3544 ExtCost += Ext0->computeCost(VF, Ctx);
3546 ExtCost += Ext1->computeCost(VF, Ctx);
3548 ExtCost += OuterExt->computeCost(VF, Ctx);
3550 return MulAccCost.
isValid() &&
3551 MulAccCost < ExtCost + MulCost + RedCost;
3556 VPValue *VecOp = Red->getVecOp();
3594 Builder.createWidenCast(Instruction::CastOps::Trunc, ValB, NarrowTy);
3596 ValB = ExtB = Builder.createWidenCast(ExtOpc, Trunc, WideTy);
3597 Mul->setOperand(1, ExtB);
3607 ExtendAndReplaceConstantOp(RecipeA, RecipeB,
B,
Mul);
3612 IsMulAccValidAndClampRange(
Mul, RecipeA, RecipeB,
nullptr)) {
3619 if (!
Sub && IsMulAccValidAndClampRange(
Mul,
nullptr,
nullptr,
nullptr))
3636 ExtendAndReplaceConstantOp(Ext0, Ext1,
B,
Mul);
3645 (Ext->getOpcode() == Ext0->getOpcode() || Ext0 == Ext1) &&
3646 Ext0->getOpcode() == Ext1->getOpcode() &&
3647 IsMulAccValidAndClampRange(
Mul, Ext0, Ext1, Ext) &&
Mul->hasOneUse()) {
3649 Ext0->getOpcode(), Ext0->getOperand(0), Ext->getScalarType(),
nullptr,
3650 *Ext0, *Ext0, Ext0->getDebugLoc());
3651 NewExt0->insertBefore(Ext0);
3656 Ext->getScalarType(),
nullptr, *Ext1,
3657 *Ext1, Ext1->getDebugLoc());
3660 auto *NewMul =
Mul->cloneWithOperands({NewExt0, NewExt1});
3661 NewMul->insertBefore(
Mul);
3662 Ext->replaceAllUsesWith(NewMul);
3663 Ext->eraseFromParent();
3664 Mul->eraseFromParent();
3678 assert(!Red->isPartialReduction() &&
3679 "This path does not support partial reductions");
3682 auto IP = std::next(Red->getIterator());
3683 auto *VPBB = Red->getParent();
3693 Red->replaceAllUsesWith(AbstractR);
3713 return CommonMetadata;
3716template <
unsigned Opcode>
3721 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
3722 "Only Load and Store opcodes supported");
3723 [[maybe_unused]]
constexpr bool IsLoad = (Opcode == Instruction::Load);
3730 for (
auto Recipes :
Groups) {
3731 if (Recipes.size() < 2)
3736 "Expected all recipes in group to have the same load-store type");
3743 VPValue *MaskI = RecipeI->getMask();
3749 bool HasComplementaryMask =
false;
3754 VPValue *MaskJ = RecipeJ->getMask();
3763 if (HasComplementaryMask) {
3764 assert(Group.
size() >= 2 &&
"must have at least 2 entries");
3774template <
typename InstType>
3792 for (
auto &Group :
Groups) {
3812 return R->isSingleScalar() == IsSingleScalar;
3814 "all members in group must agree on IsSingleScalar");
3819 LoadWithMinAlign->getUnderlyingInstr(), {EarliestLoad->getOperand(0)},
3820 IsSingleScalar,
nullptr, *EarliestLoad, CommonMetadata);
3822 UnpredicatedLoad->insertBefore(EarliestLoad);
3826 Load->replaceAllUsesWith(UnpredicatedLoad);
3827 Load->eraseFromParent();
3836 if (!StoreLoc || !StoreLoc->AATags.Scope)
3843 SinkStoreInfo SinkInfo(StoresToSink, *StoresToSink[0], PSE, L);
3855 for (
auto &Group :
Groups) {
3868 VPValue *SelectedValue = Group[0]->getOperand(0);
3871 bool IsSingleScalar = Group[0]->isSingleScalar();
3872 for (
unsigned I = 1;
I < Group.size(); ++
I) {
3873 assert(IsSingleScalar == Group[
I]->isSingleScalar() &&
3874 "all members in group must agree on IsSingleScalar");
3875 VPValue *Mask = Group[
I]->getMask();
3877 SelectedValue = Builder.createSelect(
3880 Value->getScalarType()));
3888 StoreWithMinAlign->getUnderlyingInstr(),
3889 {SelectedValue, LastStore->getOperand(1)}, IsSingleScalar,
3890 nullptr, *LastStore, CommonMetadata);
3891 UnpredicatedStore->insertBefore(*InsertBB, LastStore->
getIterator());
3895 Store->eraseFromParent();
3910 VPValue *OpV,
unsigned Idx,
bool IsScalable) {
3915 if (Member0Op == OpV)
3925 return !IsScalable && !W->getMask() && W->isConsecutive() &&
3928 return IR->getInterleaveGroup()->isFull() &&
IR->getVPValue(Idx) == OpV;
3943 if (R->getScalarType() != WideMember0->getScalarType())
3945 if (R->hasPredicate() && R->getPredicate() != WideMember0->getPredicate())
3949 for (
unsigned Idx = 0; Idx != WideMember0->getNumOperands(); ++Idx) {
3952 OpsI.
push_back(
Op->getDefiningRecipe()->getOperand(Idx));
3957 if (
any_of(
enumerate(OpsI), [WideMember0, Idx, IsScalable](
const auto &
P) {
3958 const auto &[OpIdx, OpV] =
P;
3959 return !
canNarrowLoad(WideMember0, Idx, OpV, OpIdx, IsScalable);
3970static std::optional<ElementCount>
3974 if (!InterleaveR || InterleaveR->
getMask())
3975 return std::nullopt;
3977 Type *GroupElementTy =
nullptr;
3981 return Op->getScalarType() == GroupElementTy;
3983 return std::nullopt;
3987 return Op->getScalarType() == GroupElementTy;
3989 return std::nullopt;
3993 if (IG->getFactor() != IG->getNumMembers())
3994 return std::nullopt;
4000 assert(
Size.isScalable() == VF.isScalable() &&
4001 "if Size is scalable, VF must be scalable and vice versa");
4002 return Size.getKnownMinValue();
4006 unsigned MinVal = VF.getKnownMinValue();
4008 if (IG->getFactor() == MinVal && GroupSize == GetVectorBitWidthForVF(VF))
4011 return std::nullopt;
4019 return RepR && RepR->isSingleScalar();
4033 if (V->isDefinedOutsideLoopRegions()) {
4036 return M->isDefinedOutsideLoopRegions() &&
4037 M->getScalarType() == V->getScalarType();
4039 "expected distinct loop-invariant values of matching scalar type");
4054 for (
unsigned Idx = 0,
E = WideMember0->getNumOperands(); Idx !=
E; ++Idx) {
4056 for (
VPValue *Member : Members)
4057 OpsI.
push_back(Member->getDefiningRecipe()->getOperand(Idx));
4058 WideMember0->setOperand(
4067 auto *LI =
cast<LoadInst>(LoadGroup->getInterleaveGroup()->getInsertPos());
4069 *LI, LoadGroup->getAddr(), LoadGroup->getMask(),
true,
4070 *LoadGroup, LoadGroup->getDebugLoc());
4076 assert(RepR->isSingleScalar() && RepR->getOpcode() == Instruction::Load &&
4077 "must be a single scalar load");
4078 NarrowedOps.
insert(RepR);
4083 VPValue *PtrOp = WideLoad->getAddr();
4085 PtrOp = VecPtr->getOperand(0);
4090 nullptr, {}, *WideLoad);
4091 N->insertBefore(WideLoad);
4096std::unique_ptr<VPlan>
4116 "unexpected branch-on-count");
4119 std::optional<ElementCount> VFToOptimize;
4133 if (R.mayWriteToMemory() && !InterleaveR)
4139 return any_of(V->users(), [&](VPUser *U) {
4140 auto *UR = cast<VPRecipeBase>(U);
4141 return UR->getParent()->getParent() != VectorLoop;
4158 std::optional<ElementCount> NarrowedVF =
4160 if (!NarrowedVF || (VFToOptimize && NarrowedVF != VFToOptimize))
4162 VFToOptimize = NarrowedVF;
4165 if (InterleaveR->getStoredValues().empty())
4170 auto *Member0 = InterleaveR->getStoredValues()[0];
4180 VPRecipeBase *DefR = Op.value()->getDefiningRecipe();
4183 auto *IR = dyn_cast<VPInterleaveRecipe>(DefR);
4184 return IR && IR->getInterleaveGroup()->isFull() &&
4185 IR->getVPValue(Op.index()) == Op.value();
4194 VFToOptimize->isScalable()))
4199 if (StoreGroups.empty())
4203 bool RequiresScalarEpilogue =
4214 std::unique_ptr<VPlan> NewPlan;
4216 NewPlan = std::unique_ptr<VPlan>(Plan.
duplicate());
4217 Plan.
setVF(*VFToOptimize);
4218 NewPlan->removeVF(*VFToOptimize);
4225 for (
auto *StoreGroup : StoreGroups) {
4227 NarrowedOps, Preheader);
4233 StoreGroup->getDebugLoc());
4240 Type *CanIVTy = VectorLoop->getCanonicalIVType();
4246 if (VFToOptimize->isScalable()) {
4249 Step = PHBuilder.createOverflowingOp(Instruction::Mul, {VScale,
UF},
4257 materializeVectorTripCount(Plan, VectorPH,
false,
4258 RequiresScalarEpilogue, Step);
4263 removeDeadRecipes(Plan);
4266 "All VPVectorPointerRecipes should have been removed");
4286 "Cannot handle loops with uncountable early exits");
4293 assert(RecurSplice &&
"expected FirstOrderRecurrenceSplice");
4300 if (
any_of(RecurSplice->users(),
4301 [](
VPUser *U) { return !cast<VPRecipeBase>(U)->getRegion(); }) &&
4382 {},
"vector.recur.extract.for.phi");
4385 ExitPhi->replaceUsesOfWith(ExtractR, PenultimateElement);
4399 VPValue *WidenIVCandidate = BinOp->getOperand(0);
4400 VPValue *InvariantCandidate = BinOp->getOperand(1);
4402 std::swap(WidenIVCandidate, InvariantCandidate);
4416 auto *ClonedOp = BinOp->
clone();
4417 if (ClonedOp->getOperand(0) == WidenIV) {
4418 ClonedOp->setOperand(0, ScalarIV);
4420 assert(ClonedOp->getOperand(1) == WidenIV &&
"one operand must be WideIV");
4421 ClonedOp->setOperand(1, ScalarIV);
4435 return std::nullopt;
4440 return std::nullopt;
4452 auto CheckSentinel = [&SE](
const SCEV *IVSCEV,
4453 bool UseMax) -> std::optional<APSInt> {
4455 for (
bool Signed : {
true,
false}) {
4464 return std::nullopt;
4472 PhiR->getRecurrenceKind()))
4481 VPValue *BackedgeVal = PhiR->getBackedgeValue();
4495 !
match(FindLastSelect,
4504 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression, PSE,
4509 "IVOfExpressionToSink not being an AddRec must imply "
4510 "FindLastExpression not being an AddRec.");
4519 bool UseMax = *StepDirection;
4520 std::optional<APSInt> SentinelVal = CheckSentinel(IVSCEV, UseMax);
4521 bool UseSigned = SentinelVal && SentinelVal->isSigned();
4528 if (IVOfExpressionToSink) {
4529 const SCEV *FindLastExpressionSCEV =
4531 if (std::optional<bool> NewUseMax =
4533 if (
auto NewSentinel =
4534 CheckSentinel(FindLastExpressionSCEV, *NewUseMax)) {
4537 SentinelVal = *NewSentinel;
4538 UseSigned = NewSentinel->isSigned();
4539 UseMax = *NewUseMax;
4540 IVSCEV = FindLastExpressionSCEV;
4541 IVOfExpressionToSink =
nullptr;
4551 if (AR->hasNoSignedWrap())
4553 else if (AR->hasNoUnsignedWrap())
4563 VPValue *NewFindLastSelect = BackedgeVal;
4565 if (!SentinelVal || IVOfExpressionToSink) {
4568 DebugLoc DL = FindLastSelect->getDefiningRecipe()->getDebugLoc();
4569 VPBuilder LoopBuilder(FindLastSelect->getDefiningRecipe());
4570 if (
match(FindLastSelect,
4572 SelectCond = LoopBuilder.
createNot(SelectCond);
4579 if (SelectCond !=
Cond || IVOfExpressionToSink) {
4582 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression,
4591 VPIRFlags Flags(MinMaxKind,
false,
false,
4597 NewFindLastSelect, Flags, ExitDL);
4600 VPValue *VectorRegionExitingVal = ReducedIV;
4601 if (IVOfExpressionToSink)
4602 VectorRegionExitingVal =
4604 ReducedIV, IVOfExpressionToSink);
4607 VPValue *StartVPV = PhiR->getStartValue();
4614 NewRdxResult = MiddleBuilder.
createSelect(Cmp, VectorRegionExitingVal,
4624 AnyOfPhi->insertAfter(PhiR);
4631 OrVal, VectorRegionExitingVal, StartVPV, ExitDL);
4644 PhiR->hasUsesOutsideReductionChain());
4645 NewPhiR->insertBefore(PhiR);
4646 PhiR->replaceAllUsesWith(NewPhiR);
4647 PhiR->eraseFromParent();
4654struct ReductionExtend {
4655 Type *SrcType =
nullptr;
4656 ExtendKind Kind = ExtendKind::PR_None;
4662struct ExtendedReductionOperand {
4666 ReductionExtend ExtendA, ExtendB;
4674struct VPPartialReductionChain {
4677 VPWidenRecipe *ReductionBinOp =
nullptr;
4679 ExtendedReductionOperand ExtendedOp;
4686 unsigned AccumulatorOpIdx;
4687 unsigned ScaleFactor;
4690 VPBlendRecipe *Blend =
nullptr;
4695static std::optional<unsigned>
4699 "Expected a non-normalized blend with two incoming values");
4705 return std::nullopt;
4706 return FirstIncomingHasOneUse ? 0 : 1;
4718 if (!
Op->hasOneUse() ||
4724 auto *Trunc = Builder.createWidenCast(Instruction::CastOps::Trunc,
4725 Op->getOperand(1), NarrowTy);
4727 Op->setOperand(1, Builder.createWidenCast(ExtOpc, Trunc, WideTy));
4736 auto *
Sub =
Op->getOperand(0)->getDefiningRecipe();
4738 assert(Ext->getOpcode() ==
4740 "Expected both the LHS and RHS extends to be the same");
4741 bool IsSigned = Ext->getOpcode() == Instruction::SExt;
4744 auto *FreezeX = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
X}));
4745 auto *FreezeY = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
Y}));
4746 auto *
Max = Builder.insert(
4748 {FreezeX, FreezeY}, SrcTy));
4749 auto *Min = Builder.insert(
4751 {FreezeX, FreezeY}, SrcTy));
4752 auto *AbsDiff = Builder.insert(
4755 return Builder.createWidenCast(Instruction::CastOps::ZExt, AbsDiff,
4756 Op->getScalarType());
4768 if (!
Mul->hasOneUse() ||
4769 (Ext->getOpcode() != MulLHS->getOpcode() && MulLHS != MulRHS) ||
4770 MulLHS->getOpcode() != MulRHS->getOpcode())
4773 auto *NewLHS = Builder.createWidenCast(
4774 MulLHS->getOpcode(), MulLHS->getOperand(0), Ext->getScalarType());
4775 auto *NewRHS = MulLHS == MulRHS
4777 : Builder.createWidenCast(MulRHS->getOpcode(),
4778 MulRHS->getOperand(0),
4779 Ext->getScalarType());
4780 auto *NewMul =
Mul->cloneWithOperands({NewLHS, NewRHS});
4781 Builder.insert(NewMul);
4782 Op->replaceAllUsesWith(NewMul);
4783 Op->eraseFromParent();
4784 Mul->eraseFromParent();
4793 VPValue *VecOp = Red->getVecOp();
4847static void transformToPartialReduction(
const VPPartialReductionChain &Chain,
4855 WidenRecipe->
getOperand(1 - Chain.AccumulatorOpIdx));
4858 ExtendedOp = optimizeExtendsForPartialReduction(ExtendedOp);
4874 if ((WidenRecipe->
getOpcode() == Instruction::Sub &&
4876 (WidenRecipe->
getOpcode() == Instruction::FSub &&
4881 if (WidenRecipe->
getOpcode() == Instruction::FSub) {
4893 Builder.insert(NegRecipe);
4894 ExtendedOp = NegRecipe;
4909 std::optional<unsigned> BlendReductionIdx =
4910 getBlendReductionUpdateValueIdx(Chain.Blend);
4911 assert(BlendReductionIdx &&
4913 "Expected blend to contain the reduction update");
4930 assert((!ExitValue || IsLastInChain) &&
4931 "if we found ExitValue, it must match RdxPhi's backedge value");
4942 PartialRed->insertBefore(WidenRecipe);
4952 E->insertBefore(WidenRecipe);
4953 PartialRed->replaceAllUsesWith(
E);
4966 auto *NewScaleFactor = Plan.
getConstantInt(32, Chain.ScaleFactor);
4967 StartInst->setOperand(2, NewScaleFactor);
4975 VPValue *OldStartValue = StartInst->getOperand(0);
4976 StartInst->setOperand(0, StartInst->getOperand(1));
4980 assert(RdxResult &&
"Could not find reduction result");
4983 unsigned SubOpc = Chain.RK ==
RecurKind::FSub ? Instruction::BinaryOps::FSub
4984 : Instruction::BinaryOps::Sub;
4990 [&NewResult](
VPUser &U,
unsigned Idx) {
return &
U != NewResult; });
4996 const VPPartialReductionChain &Link,
4999 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5000 std::optional<unsigned> BinOpc = std::nullopt;
5002 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5003 BinOpc = ExtendedOp.ExtendsUser->
getOpcode();
5005 std::optional<llvm::FastMathFlags>
Flags;
5009 auto GetLinkOpcode = [&Link]() ->
unsigned {
5012 return Instruction::Add;
5014 return Instruction::FAdd;
5016 return Link.ReductionBinOp->
getOpcode();
5021 GetLinkOpcode(), ExtendedOp.ExtendA.SrcType, ExtendedOp.ExtendB.SrcType,
5022 RdxType, VF, ExtendedOp.ExtendA.Kind, ExtendedOp.ExtendB.Kind, BinOpc,
5043static std::optional<ExtendedReductionOperand>
5046 "Op should be operand of UpdateR");
5054 if (
Op->hasOneUse() &&
5063 Type *RHSInputType =
Y->getScalarType();
5064 if (LHSInputType != RHSInputType ||
5065 LHSExt->getOpcode() != RHSExt->getOpcode())
5066 return std::nullopt;
5069 return ExtendedReductionOperand{
5071 {LHSInputType, getPartialReductionExtendKind(LHSExt)},
5075 std::optional<TTI::PartialReductionExtendKind> OuterExtKind;
5078 VPValue *CastSource = CastRecipe->getOperand(0);
5079 OuterExtKind = getPartialReductionExtendKind(CastRecipe);
5089 return ExtendedReductionOperand{
5096 if (!
Op->hasOneUse())
5097 return std::nullopt;
5102 return std::nullopt;
5112 return std::nullopt;
5116 ExtendKind LHSExtendKind = getPartialReductionExtendKind(LHSCast);
5119 const APInt *RHSConst =
nullptr;
5125 return std::nullopt;
5129 if (Cast && OuterExtKind &&
5130 getPartialReductionExtendKind(Cast) != OuterExtKind)
5131 return std::nullopt;
5133 Type *RHSInputType = LHSInputType;
5134 ExtendKind RHSExtendKind = LHSExtendKind;
5137 RHSExtendKind = getPartialReductionExtendKind(RHSCast);
5140 return ExtendedReductionOperand{
5141 MulOp, {LHSInputType, LHSExtendKind}, {RHSInputType, RHSExtendKind}};
5148static std::optional<SmallVector<VPPartialReductionChain>>
5155 return std::nullopt;
5165 VPValue *CurrentValue = ExitValue;
5166 while (CurrentValue != RedPhiR) {
5168 std::optional<unsigned> BlendReductionIdx;
5172 return std::nullopt;
5174 BlendReductionIdx = getBlendReductionUpdateValueIdx(Blend);
5175 if (!BlendReductionIdx)
5176 return std::nullopt;
5183 return std::nullopt;
5190 std::optional<ExtendedReductionOperand> ExtendedOp =
5191 matchExtendedReductionOperand(UpdateR,
Op);
5193 ExtendedOp = matchExtendedReductionOperand(UpdateR, PrevValue);
5195 return std::nullopt;
5203 return std::nullopt;
5205 Type *ExtSrcType = ExtendedOp->ExtendA.SrcType;
5208 return std::nullopt;
5210 VPPartialReductionChain Link(
5211 {UpdateR, *ExtendedOp, RK,
5216 CurrentValue = PrevValue;
5221 std::reverse(Chain.
begin(), Chain.
end());
5240 if (
auto Chains = getScaledReductions(RedPhiR))
5241 ChainsByPhi.
try_emplace(RedPhiR, std::move(*Chains));
5244 if (ChainsByPhi.
empty())
5252 for (
const auto &[
_, Chains] : ChainsByPhi)
5253 for (
const VPPartialReductionChain &Chain : Chains) {
5254 PartialReductionOps.
insert(Chain.ExtendedOp.ExtendsUser);
5256 PartialReductionBlends.
insert(Chain.Blend);
5257 ScaledReductionMap[Chain.ReductionBinOp] = Chain.ScaleFactor;
5263 auto ExtendUsersValid = [&](
VPValue *Ext) {
5265 return PartialReductionOps.contains(cast<VPRecipeBase>(U));
5269 auto IsProfitablePartialReductionChainForVF =
5276 for (
const VPPartialReductionChain &Link : Chain) {
5277 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5278 InstructionCost LinkCost = getPartialReductionLinkCost(CostCtx, Link, VF);
5282 PartialCost += LinkCost;
5283 RegularCost += Link.ReductionBinOp->
computeCost(VF, CostCtx);
5285 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5286 RegularCost += ExtendedOp.ExtendsUser->
computeCost(VF, CostCtx);
5289 RegularCost += Extend->computeCost(VF, CostCtx);
5291 return PartialCost.
isValid() && PartialCost < RegularCost;
5299 for (
auto &[RedPhiR, Chains] : ChainsByPhi) {
5300 for (
const VPPartialReductionChain &Chain : Chains) {
5301 if (!
all_of(Chain.ExtendedOp.ExtendsUser->operands(), ExtendUsersValid)) {
5305 auto UseIsValid = [&, RedPhiR = RedPhiR](
VPUser *U) {
5307 return PhiR == RedPhiR;
5311 return Blend == Chain.Blend || PartialReductionBlends.
contains(Blend);
5313 return Chain.ScaleFactor == ScaledReductionMap.
lookup_or(R, 0) ||
5319 if (!
all_of(Chain.ReductionBinOp->users(), UseIsValid)) {
5328 auto *RepR = dyn_cast<VPReplicateRecipe>(U);
5329 return RepR && RepR->getOpcode() == Instruction::Store;
5340 return IsProfitablePartialReductionChainForVF(Chains, VF);
5346 for (
auto &[Phi, Chains] : ChainsByPhi)
5347 for (
const VPPartialReductionChain &Chain : Chains)
5348 transformToPartialReduction(Chain, Plan, Phi);
5363 if (VPI && VPI->getUnderlyingValue() &&
5374 auto ProcessSubset = [&](
VPlan &,
auto ProcessVPInst) {
5377 if (!ProcessVPInst(VPI))
5386 assert(New->getParent() &&
"New recipe must have been inserted");
5387 if (VPI->
getOpcode() == Instruction::Load)
5396 return ReplaceWith(VPI,
VPBuilder(VPI).insert(
5403 "lowerMemoryIdioms", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5405 VPI, FinalRedStoresBuilder))
5414 return ReplaceWith(VPI,
VPBuilder(VPI).insert(Histogram));
5427 "scalarizeMemOpsWithIrregularTypes", ProcessSubset, Plan,
5431 return Scalarize(VPI);
5438 "makeVPlanMemOpDecision", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5440 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5450 const SCEV *PtrSCEV =
5452 bool IsSingleScalarLoad =
5458 I, Ptr, IsSingleScalarLoad,
5467 "widenConsecutiveMemOps", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5469 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5473 std::optional<int64_t> Stride =
5475 if (Stride != 1 && Stride != -1)
5506 return ReplaceWith(VPI,
Load);
5515 auto *StoreR = Builder.createWidenStore(
5518 return ReplaceWith(VPI, StoreR);
5525 return ReplaceWith(VPI, Recipe);
5527 return Scalarize(VPI);
5550 if (VPI->mayHaveSideEffects())
5554 if (VPI->isMasked() && !VPI->isSafeToSpeculativelyExecute())
5559 if (VPI->getOpcode() == Instruction::Add &&
5568 VPI->getOpcode(), VPI->operandsWithoutMask(),
nullptr, *VPI,
5569 *VPI, VPI->getDebugLoc(),
I);
5570 Recipe->insertBefore(VPI);
5571 VPI->replaceAllUsesWith(Recipe);
5572 VPI->eraseFromParent();
5582 switch (Param.ParamKind) {
5583 case VFParamKind::Vector:
5584 case VFParamKind::GlobalPredicate:
5586 case VFParamKind::OMP_Uniform:
5587 return SE->isSCEVable(Args[Param.ParamPos]->getScalarType()) &&
5588 SE->isLoopInvariant(
5589 vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5591 case VFParamKind::OMP_Linear:
5592 return match(vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5593 m_scev_AffineAddRec(
5594 m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
5595 m_SpecificLoop(L)));
5612 const auto *It =
find_if(Mappings, [&](
const VFInfo &Info) {
5613 return Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()) &&
5616 if (It == Mappings.end())
5623struct CallWideningDecision {
5624 enum class KindTy { Scalarize,
Intrinsic, VectorVariant };
5625 CallWideningDecision(KindTy Kind,
Function *Variant =
nullptr)
5648 return CallWideningDecision::KindTy::Scalarize;
5658 return CallWideningDecision::KindTy::Scalarize;
5662 false, VF, CostCtx);
5677 return CallWideningDecision::KindTy::Intrinsic;
5681 if (VecFunc && ScalarCost >= VecCallCost)
5682 return {CallWideningDecision::KindTy::VectorVariant, VecFunc};
5684 return CallWideningDecision::KindTy::Scalarize;
5694 if (!VPI || !VPI->getUnderlyingValue() ||
5695 VPI->getOpcode() != Instruction::Call)
5700 VPI->op_begin() + CI->arg_size());
5702 CallWideningDecision Decision =
5711 switch (Decision.Kind) {
5712 case CallWideningDecision::KindTy::Intrinsic: {
5716 *VPI, VPI->getDebugLoc());
5719 case CallWideningDecision::KindTy::VectorVariant: {
5723 VPValue *Mask = VPI->isMasked() ? VPI->getMask() : Plan.
getTrue();
5724 Ops.push_back(Mask);
5726 Ops.push_back(VPI->getOperand(VPI->getNumOperandsWithoutMask() - 1));
5728 *VPI, VPI->getDebugLoc());
5731 case CallWideningDecision::KindTy::Scalarize:
5737 VPI->replaceAllUsesWith(Replacement);
5738 VPI->eraseFromParent();
5760 if (!MemR || MemR->isConsecutive())
5763 VPValue *Ptr = MemR->getAddr();
5775 VPValue *StoredValue =
nullptr;
5779 StoredValue = StoreR->getStoredValue();
5781 IntrinID = Intrinsic::experimental_vp_strided_store;
5785 IntrinID = Intrinsic::experimental_vp_strided_load;
5788 Align Alignment = MemR->getAlign();
5791 if (!Ctx.TTI.isLegalStridedLoadStore(VectorTy, Alignment))
5796 IntrinID, VectorTy, MemR->isMasked(), Alignment, Ctx);
5797 return StridedLoadStoreCost < CurrentCost;
5808 Ctx.invalidateWideningDecision(&MemR->getIngredient(), VF);
5813 I32VF = Builder.createScalarZExtOrTrunc(
5827 "Stride type from SCEV must match the index type");
5828 VPValue *CanIV = Builder.createScalarZExtOrTrunc(
5831 auto *
Offset = Builder.createOverflowingOp(
5832 Instruction::Mul, {CanIV, StrideInBytes},
5833 {AddRecPtr->hasNoUnsignedWrap(),
false});
5837 VPValue *BasePtr = Builder.createNoWrapPtrAdd(StartVPV,
Offset, NWFlags);
5840 VPValue *NewPtr = Builder.createVectorPointer(
5844 VPValue *Mask = MemR->getMask();
5849 Ops.push_back(StoredValue);
5850 Ops.append({NewPtr, StrideInBytes, Mask, I32VF});
5852 auto *StridedR = Builder.createWidenMemIntrinsic(
5855 *MemR, R.getDebugLoc());
5858 R.eraseFromParent();
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static bool isEqual(const Function &Caller, const Function &Callee)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static cl::opt< IntrinsicCostStrategy > IntrinsicCost("intrinsic-cost-strategy", cl::desc("Costing strategy for intrinsic instructions"), cl::init(IntrinsicCostStrategy::InstructionCost), cl::values(clEnumValN(IntrinsicCostStrategy::InstructionCost, "instruction-cost", "Use TargetTransformInfo::getInstructionCost"), clEnumValN(IntrinsicCostStrategy::IntrinsicCost, "intrinsic-cost", "Use TargetTransformInfo::getIntrinsicInstrCost"), clEnumValN(IntrinsicCostStrategy::TypeBasedIntrinsicCost, "type-based-intrinsic-cost", "Calculate the intrinsic cost based only on argument types")))
iv Induction Variable Users
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
Legalize the Machine IR a function s Machine IR
This file provides utility analysis objects describing memory locations.
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
This file builds on the ADT/GraphTraits.h file to build a generic graph post order iterator.
const SmallVectorImpl< MachineOperand > & Cond
This is the interface for a metadata-based scoped no-alias analysis.
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
This file implements the TypeSwitch template, which mimics a switch() statement whose cases are type ...
This file implements dominator tree analysis for a single level of a VPlan's H-CFG.
This file contains the declarations of different VPlan-related auxiliary helpers.
This file contains the declarations of the Vectorization Plan base classes:
static const X86InstrFMA3Group Groups[]
static const uint32_t IV[8]
Helper for extra no-alias checks via known-safe recipe and SCEV.
SinkStoreInfo(ArrayRef< VPReplicateRecipe * > ExcludeRecipes, VPReplicateRecipe &GroupLeader, PredicatedScalarEvolution &PSE, const Loop &L)
SinkStoreInfo(VPReplicateRecipe &GroupLeader)
bool shouldSkip(VPRecipeBase &R) const
Return true if R should be skipped during alias checking, either because it's in the exclude set or b...
Class for arbitrary precision integers.
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
unsigned getActiveBits() const
Compute the number of active bits in the value.
APInt abs() const
Get the absolute value.
unsigned getBitWidth() const
Return the number of bits in the APInt.
int32_t exactLogBase2() const
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
An arbitrary precision integer that knows its signedness.
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
@ NoAlias
The two locations do not alias at all.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & back() const
Get the last element.
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
const T & front() const
Get the first element.
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
const Function * getParent() const
Return the enclosing method, or null if none.
bool isNoBuiltin() const
Return true if the call should not be treated as a call to a builtin.
This class represents a function call, abstracting a target machine's calling convention.
@ ICMP_ULT
unsigned less than
@ ICMP_ULE
unsigned less or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
This class represents a range of values.
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
A parsed version of the target data layout string in and methods for querying it.
LLVM_ABI IntegerType * getIndexType(LLVMContext &C, unsigned AddressSpace) const
Returns the type of a GEP index in AddressSpace.
static DebugLoc getUnknown()
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
ValueT lookup_or(const_arg_type_t< KeyT > Val, U &&Default) const
bool dominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
dominates - Returns true iff A dominates B.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
static constexpr ElementCount getScalable(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
Convenience struct for specifying and reasoning about fast-math flags.
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags noUnsignedWrap()
bool hasNoUnsignedWrap() const
GEPNoWrapFlags withoutNoUnsignedWrap() const
static GEPNoWrapFlags none()
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
A struct for saving information about induction variables.
InductionKind
This enum represents the kinds of inductions that we support.
@ IK_PtrInduction
Pointer induction var. Step = C.
@ IK_IntInduction
Integer induction variable. Step = C.
static InstructionCost getInvalid(CostType Val=0)
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
The group of interleaved loads/stores sharing the same stride and close to each other.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
static bool getDecisionAndClampRange(const std::function< bool(ElementCount)> &Predicate, VFRange &Range)
Test a Predicate on a Range of VF's.
Represents a single loop in the control flow graph.
This class implements a map that also provides access to all stored values in a deterministic order.
ValueT lookup(const KeyT &Key) const
std::pair< iterator, bool > try_emplace(const KeyT &Key, Ts &&...Args)
Representation for a specific memory location.
Function * getFunction(StringRef Name) const
Look up the specified function in the module symbol table.
Post-order traversal of a graph.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
ScalarEvolution * getSE() const
Returns the ScalarEvolution analysis used.
LLVM_ABI const SCEV * getSCEV(Value *V)
Returns the SCEV expression of V, in the context of the current SCEV predicate.
static LLVM_ABI unsigned getOpcode(RecurKind Kind)
Returns the opcode corresponding to the RecurrenceKind.
unsigned getOpcode() const
static bool isFindLastRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
RegionT * getParent() const
Get the parent of the Region.
This class represents a constant integer value.
ConstantInt * getValue() const
static const SCEV * rewrite(const SCEV *Scev, ScalarEvolution &SE, ValueToSCEVMapTy &Map)
This means that we are dealing with an entirely unknown SCEV value, and only represent it as its LLVM...
This class represents an analyzed expression in the program.
Type * getType() const
Return the LLVM type of this SCEV expression.
The main scalar evolution driver.
const DataLayout & getDataLayout() const
Return the DataLayout associated with the module this SCEV instance is operating on.
LLVM_ABI const SCEV * getNegativeSCEV(const SCEV *V, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
Return the SCEV object corresponding to -V.
LLVM_ABI bool isKnownNegative(const SCEV *S)
Test if the given expression is known to be negative.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Return LHS-RHS.
ConstantRange getSignedRange(const SCEV *S)
Determine the signed range for a particular SCEV.
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
LLVM_ABI bool isKnownPositive(const SCEV *S)
Test if the given expression is known to be positive.
LLVM_ABI const SCEV * getElementCount(Type *Ty, ElementCount EC, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
ConstantRange getUnsignedRange(const SCEV *S)
Determine the unsigned range for a particular SCEV.
LLVM_ABI bool isKnownPredicate(CmpPredicate Pred, SCEVUse LHS, SCEVUse RHS)
Test if the given expression is known to satisfy the condition described by Pred, LHS,...
static LLVM_ABI AliasResult alias(const MemoryLocation &LocA, const MemoryLocation &LocB)
A vector that has set insertion semantics.
size_type size() const
Determine the number of elements in the SetVector.
bool insert(const value_type &X)
Insert a new element into the SetVector.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Provides information about what library functions are available for the current target.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
This class implements a switch-like dispatch statement for a value of 'T' using dyn_cast functionalit...
TypeSwitch< T, ResultT > & Case(CallableT &&caseFn)
Add a case on the given type.
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntOrPtrTy() const
Return true if this is an integer type or a pointer type.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static SmallVector< VFInfo, 8 > getMappings(const CallInst &CI)
Retrieve all the VFInfo instances associated to the CallInst CI.
bool isLegalMaskedLoadOrStore(bool IsLoad, Type *ScalarTy, Align Alignment, unsigned AddressSpace) const
Returns true if the target machine supports a masked load (if IsLoad) or masked store of scalar type ...
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
void appendRecipe(VPRecipeBase *Recipe)
Augment the existing recipes of a VPBasicBlock with an additional Recipe as the last recipe.
iterator begin()
Recipe iterator methods.
iterator_range< iterator > phis()
Returns an iterator range over the PHI-like recipes in the block.
iterator getFirstNonPhi()
Return the position of the first non-phi node recipe in the block.
VPBasicBlock * splitAt(iterator SplitAt)
Split current block at SplitAt by inserting a new block between the current block and its successors ...
const VPRecipeBase & front() const
VPRecipeBase * getTerminator()
If the block has multiple successors, return the branch recipe terminating the block.
const VPRecipeBase & back() const
A recipe for vectorizing a phi-node as a sequence of mask-based select instructions.
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
VPValue * getMask(unsigned Idx) const
Return mask number Idx.
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
void setMask(unsigned Idx, VPValue *V)
Set mask number Idx to V.
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
void setSuccessors(ArrayRef< VPBlockBase * > NewSuccs)
Set each VPBasicBlock in NewSuccss as successor of this VPBlockBase.
VPRegionBlock * getParent()
const VPBasicBlock * getExitingBasicBlock() const
size_t getNumSuccessors() const
void setPredecessors(ArrayRef< VPBlockBase * > NewPreds)
Set each VPBasicBlock in NewPreds as predecessor of this VPBlockBase.
const VPBlocksTy & getPredecessors() const
VPBlockBase * getSinglePredecessor() const
const VPBasicBlock * getEntryBasicBlock() const
VPBlockBase * getSingleSuccessor() const
const VPBlocksTy & getSuccessors() const
static auto blocksAs(T &&Range)
Return an iterator range over Range with each block cast to BlockTy.
static void insertOnEdge(VPBlockBase *From, VPBlockBase *To, VPBlockBase *BlockPtr)
Inserts BlockPtr on the edge between From and To.
static bool isLatch(const VPBlockBase *VPB, const VPDominatorTree &VPDT)
Returns true if VPB is a loop latch, using isHeader().
static VPBasicBlock * getPlainCFGMiddleBlock(const VPlan &Plan)
Returns the middle block of Plan in plain CFG form (before regions are formed).
static void insertTwoBlocksAfter(VPBlockBase *IfTrue, VPBlockBase *IfFalse, VPBlockBase *BlockPtr)
Insert disconnected VPBlockBases IfTrue and IfFalse after BlockPtr.
static void connectBlocks(VPBlockBase *From, VPBlockBase *To, unsigned PredIdx=-1u, unsigned SuccIdx=-1u)
Connect VPBlockBases From and To bi-directionally.
static void disconnectBlocks(VPBlockBase *From, VPBlockBase *To)
Disconnect VPBlockBases From and To bi-directionally.
static auto blocksOnly(T &&Range)
Return an iterator range over Range which only includes BlockTy blocks.
static std::pair< VPBasicBlock *, VPBasicBlock * > getPlainCFGHeaderAndLatch(const VPlan &Plan)
Returns the header and latch of the outermost loop of Plan in plain CFG form (before regions are form...
static void transferSuccessors(VPBlockBase *Old, VPBlockBase *New)
Transfer successors from Old to New. New must have no successors.
static SmallVector< VPBasicBlock * > blocksInSingleSuccessorChainBetween(VPBasicBlock *FirstBB, VPBasicBlock *LastBB)
Returns the blocks between FirstBB and LastBB, where FirstBB to LastBB forms a single-sucessor chain.
A recipe for generating conditional branches on the bits of a mask.
VPlan-based builder utility analogous to IRBuilder.
VPInstruction * createFirstActiveLane(ArrayRef< VPValue * > Masks, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenStoreRecipe * createWidenStore(StoreInst &Store, VPValue *Addr, VPValue *StoredVal, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Store, storing StoredVal to Addr with Mask (may be null).
VPInstruction * createAdd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", VPRecipeWithIRFlags::WrapFlagsTy WrapFlags={false, false})
VPInstruction * createOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createLogicalOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenLoadRecipe * createWidenLoad(LoadInst &Load, VPValue *Addr, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Load, loading from Addr with Mask (may be null).
VPInstruction * createNot(VPValue *Operand, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createAnyOfReduction(VPValue *ChainOp, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown())
Create an AnyOf reduction pattern: or-reduce ChainOp, freeze the result, then select between TrueVal ...
void setInsertPoint(const VPInsertPoint &IP)
Set the current insert point.
VPInstruction * createLogicalAnd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createScalarCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy, DebugLoc DL, std::optional< VPIRFlags > Flags=std::nullopt, const VPIRMetadata &Metadata={})
VPValue * createScalarZExtOrTrunc(VPValue *Op, Type *ResultTy, DebugLoc DL)
static VPBuilder getToInsertAfter(VPRecipeBase *R)
Create a VPBuilder to insert after R.
VPDerivedIVRecipe * createDerivedIV(InductionDescriptor::InductionKind Kind, FPMathOperator *FPBinOp, VPValue *Start, VPValue *Current, VPValue *Step, const VPIRFlags::WrapFlagsTy &Flags={})
Convert Current to Start + Current * Step.
VPWidenCastRecipe * createWidenCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy)
VPInstruction * createICmp(CmpInst::Predicate Pred, VPValue *A, VPValue *B, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
Create a new ICmp VPInstruction with predicate Pred and operands A and B.
VPInstruction * createSelect(VPValue *Cond, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", std::optional< VPIRFlags > Flags=std::nullopt)
Create a select of TrueVal and FalseVal based on Cond, using the default flags for the result type,...
VPInstruction * createNaryOp(unsigned Opcode, ArrayRef< VPValue * > Operands, Instruction *Inst=nullptr, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
Create an N-ary operation with Opcode, Operands and set Inst as its underlying Instruction.
static VPSingleDefRecipe * createSingleScalarOp(unsigned Opcode, ArrayRef< VPValue * > Operands, VPValue *Mask, const VPIRFlags &Flags, const VPIRMetadata &Metadata, DebugLoc DL, Instruction *UV)
Create a single-scalar recipe with Opcode and Operands without inserting it.
unsigned getNumDefinedValues() const
Returns the number of values defined by the VPDef.
VPValue * getVPSingleValue()
Returns the only VPValue defined by the VPDef.
VPValue * getVPValue(unsigned I)
Returns the VPValue with index I defined by the VPDef.
ArrayRef< VPRecipeValue * > definedValues()
Returns an ArrayRef of the values defined by the VPDef.
Template specialization of the standard LLVM dominator tree utility for VPBlockBases.
bool properlyDominates(const VPRecipeBase *A, const VPRecipeBase *B) const
A recipe to combine multiple recipes into a single 'expression' recipe, which should be considered a ...
A recipe representing a sequence of load -> update -> store as part of a histogram operation.
A special type of VPBasicBlock that wraps an existing IR basic block.
Class to record and manage LLVM IR flags.
static VPIRFlags getDefaultFlags(unsigned Opcode, Type *ResultTy=nullptr)
Returns default flags for Opcode and scalar ResultTy for opcodes that support it, asserts otherwise.
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
This is a concrete Recipe that models a single VPlan-level instruction.
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the VPInstruction is masked.
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
@ ExtractPenultimateElement
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
@ BuildVector
Creates a fixed-width vector containing all operands.
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
unsigned getOpcode() const
VPValue * getMask() const
Returns the mask for the VPInstruction.
const InterleaveGroup< Instruction > * getInterleaveGroup() const
VPValue * getMask() const
Return the mask used by this recipe.
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
VPInterleaveRecipe is a recipe for transforming an interleave group of load or stores into one wide l...
VPPredInstPHIRecipe is a recipe for generating the phi nodes needed when control converges back from ...
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
VPRegionBlock * getRegion()
VPBasicBlock * getParent()
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
Helper class to create VPRecipies from IR instructions.
VPHistogramRecipe * widenIfHistogram(VPInstruction *VPI)
If VPI represents a histogram operation (as determined by LoopVectorizationLegality) make that safe f...
bool prefersVectorizedAddressing() const
Returns true if the target prefers vectorized addressing.
VPRecipeBase * tryToWidenMemory(VPInstruction *VPI, VFRange &Range)
Check if the load or store instruction VPI should widened for Range.Start and potentially masked.
bool replaceWithFinalIfReductionStore(VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder)
If VPI is a store of a reduction into an invariant address, delete it.
VPSingleDefRecipe * handleReplication(VPInstruction *VPI, VFRange &Range)
Build a replicating or single-scalar recipe for VPI.
bool isPredicatedInst(Instruction *I) const
Returns true if I needs to be predicated (i.e.
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
A recipe for handling reduction phis.
void setVFScaleFactor(unsigned ScaleFactor)
Set the VFScaleFactor for this reduction phi.
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
RecurKind getRecurrenceKind() const
Returns the recurrence kind of the reduction.
A recipe to represent inloop, ordered or partial reduction operations.
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
const VPBlockBase * getEntry() const
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
void setExiting(VPBlockBase *ExitingBlock)
Set ExitingBlock as the exiting VPBlockBase of this VPRegionBlock.
Type * getCanonicalIVType() const
Return the type of the canonical IV for loop regions.
VPRegionValue * getCanonicalIV()
Return the canonical induction variable of the region, null for replicating regions.
const VPBlockBase * getExiting() const
VPRegionValue * getHeaderMask() const
Return the header mask of the region, or null if not set.
VPReplicateRecipe replicates a given instruction producing multiple scalar copies of the original sca...
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy, ArrayRef< const VPValue * > ArgOps, bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx)
Return the cost of scalarizing a call to CalledFn with argument operands ArgOps for a given VF.
operand_range operandsWithoutMask()
Return the recipe's operands, excluding the mask of a predicated recipe.
bool isPredicated() const
VPValue * getMask()
Return the mask of a predicated VPReplicateRecipe.
Lightweight SCEV-to-VPlan expander.
VPValue * expand(const SCEV *S)
Expand S into recipes and live-ins using the builder.
A recipe for handling phi nodes of integer and floating-point inductions, producing their scalar valu...
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
VPSingleDefRecipe * clone() override=0
Clone the current recipe.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
void setOperand(unsigned I, VPValue *New)
unsigned getNumOperands() const
VPValue * getOperand(unsigned N) const
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
bool isDefinedOutsideLoopRegions() const
Returns true if the VPValue is defined outside any loop.
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
bool hasMoreThanOneUniqueUser() const
Returns true if the value has more than one unique user.
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
VPUser * getSingleUser()
Return the single user of this value, or nullptr if there is not exactly one user.
void replaceAllUsesWith(VPValue *New)
void replaceUsesWithIf(VPValue *New, llvm::function_ref< bool(VPUser &U, unsigned Idx)> ShouldReplace)
Go through the uses list for this VPValue and make each use point to New if the callback ShouldReplac...
A recipe to compute a pointer to the last element of each part of a widened memory access for widened...
A recipe for widening Call instructions using library calls.
static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx)
Return the cost of widening a call using the vector function Variant.
VPWidenCastRecipe is a recipe to create vector cast instructions.
Instruction::CastOps getOpcode() const
A recipe for handling GEP instructions.
Base class for widened induction (VPWidenIntOrFpInductionRecipe and VPWidenPointerInductionRecipe),...
VPIRValue * getStartValue() const
Returns the start value of the induction.
PHINode * getPHINode() const
Returns the underlying PHINode if one exists, or null otherwise.
VPValue * getStepValue()
Returns the step value of the induction.
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
A recipe for handling phi nodes of integer and floating-point inductions, producing their vector valu...
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
A recipe for widening vector intrinsics.
static InstructionCost computeCallCost(Intrinsic::ID ID, ArrayRef< const VPValue * > Operands, const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx)
Compute the cost of a vector intrinsic with ID and Operands.
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
A common mixin class for widening memory operations.
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
A recipe for widened phis.
VPWidenRecipe is a recipe for producing a widened instruction using the opcode and operands of the re...
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenRecipe.
VPWidenRecipe * clone() override
Clone the current recipe.
unsigned getOpcode() const
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
VPIRValue * getLiveIn(Value *V) const
Return the live-in VPIRValue for V, if there is one or nullptr otherwise.
bool hasVF(ElementCount VF) const
const DataLayout & getDataLayout() const
LLVMContext & getContext() const
VPBasicBlock * getEntry()
bool hasScalableVF() const
VPValue * getTripCount() const
The trip count of the original loop.
VPValue * getOrCreateBackedgeTakenCount()
The backedge taken count of the original loop.
iterator_range< SmallSetVector< ElementCount, 2 >::iterator > vectorFactors() const
Returns an iterator range over all VFs of the plan.
VPIRValue * getFalse()
Return a VPIRValue wrapping i1 false.
VPSymbolicValue & getVFxUF()
Returns VF * UF of the vector loop region.
VPIRValue * getAllOnesValue(Type *Ty)
Return a VPIRValue wrapping the AllOnes value of type Ty.
VPRegionBlock * createReplicateRegion(VPBlockBase *Entry, VPBlockBase *Exiting, const std::string &Name="")
Create a new replicate region with Entry, Exiting and Name.
auto getLiveIns() const
Return the list of live-in VPValues available in the VPlan.
bool hasUF(unsigned UF) const
ArrayRef< VPIRBasicBlock * > getExitBlocks() const
Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of the original scalar loop.
VPSymbolicValue & getVectorTripCount()
The vector trip count.
VPValue * getBackedgeTakenCount() const
VPIRValue * getOrAddLiveIn(Value *V)
Gets the live-in VPIRValue for V or adds a new live-in (if none exists yet) for V.
VPIRValue * getZero(Type *Ty)
Return a VPIRValue wrapping the null value of type Ty.
void setVF(ElementCount VF)
bool isUnrolled() const
Returns true if the VPlan already has been unrolled, i.e.
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
unsigned getConcreteUF() const
Returns the concrete UF of the plan, after unrolling.
void resetTripCount(VPValue *NewTripCount)
Resets the trip count for the VPlan.
VPBasicBlock * getMiddleBlock()
Returns the 'middle' block of the plan, that is the block that selects whether to execute the scalar ...
VPBasicBlock * createVPBasicBlock(const Twine &Name, VPRecipeBase *Recipe=nullptr)
Create a new VPBasicBlock with Name and containing Recipe if present.
VPIRValue * getTrue()
Return a VPIRValue wrapping i1 true.
VPBasicBlock * getVectorPreheader() const
Returns the preheader of the vector loop region, if one exists, or null otherwise.
VPSymbolicValue & getUF()
Returns the UF of the vector loop region.
bool hasScalarVFOnly() const
VPBasicBlock * getScalarPreheader() const
Return the VPBasicBlock for the preheader of the scalar loop.
bool hasTailFolded() const
Returns true if the vector loop region is tail-folded.
VPSymbolicValue & getVF()
Returns the VF of the vector loop region.
LLVM_ABI_FOR_TEST VPlan * duplicate()
Clone the current VPlan, update all VPValues of the new VPlan and cloned recipes to refer to the clon...
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
LLVM Value Representation.
iterator_range< user_iterator > users()
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
constexpr bool hasKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns true if there exists a value X where RHS.multiplyCoefficientBy(X) will result in a value whos...
constexpr ScalarTy getFixedValue() const
constexpr ScalarTy getKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns a value X where RHS.multiplyCoefficientBy(X) will result in a value whose quantity matches ou...
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr LeafTy multiplyCoefficientBy(ScalarTy RHS) const
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
An efficient, type-erasing, non-owning reference to a callable.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
std::variant< std::monostate, Loc::Single, Loc::Multi, Loc::MMI, Loc::EntryValue > Variant
Alias for the std::variant specialization base class of DbgVariable.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
AllOnesConstantMatch m_AllOnes()
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_unless< Pattern > m_Unless(const Pattern &P)
Match if the inner matcher does NOT match.
match_isa< To... > m_Isa()
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::URem > m_URem(const LHS &L, const RHS &R)
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
SpecificCmpClass_match< LHS, RHS, CmpInst > m_SpecificCmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::UDiv > m_UDiv(const LHS &L, const RHS &R)
SelectLike_match< CondTy, LTy, RTy > m_SelectLike(const CondTy &C, const LTy &TrueC, const RTy &FalseC)
Matches a value that behaves like a boolean-controlled select, i.e.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
CastOperator_match< OpTy, Instruction::BitCast > m_BitCast(const OpTy &Op)
Matches BitCast.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinaryOp_match< LHS, RHS, Instruction::FAdd, true > m_c_FAdd(const LHS &L, const RHS &R)
Matches FAdd with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And, true > m_c_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R with LHS and RHS in either order.
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
bind_cst_ty m_scev_APInt(const APInt *&C)
Match an SCEV constant and bind it to an APInt.
specificloop_ty m_SpecificLoop(const Loop *L)
bool match(const SCEV *S, const Pattern &P)
SCEVAffineAddRec_match< Op0_t, Op1_t, match_isa< const Loop > > m_scev_AffineAddRec(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > > m_ExtractLastLaneOfLastPart(const Op0_t &Op0)
AllRecipe_commutative_match< Instruction::And, Op0_t, Op1_t > m_c_BinaryAnd(const Op0_t &Op0, const Op1_t &Op1)
Match a binary AND operation.
AllRecipe_match< Instruction::Or, Op0_t, Op1_t > m_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
Match a binary OR operation.
VPInstruction_match< VPInstruction::AnyOf > m_AnyOf()
AllRecipe_commutative_match< Instruction::Or, Op0_t, Op1_t > m_c_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ComputeReductionResult, Op0_t > m_ComputeReductionResult(const Op0_t &Op0)
auto m_WidenAnyExtend(const Op0_t &Op0)
match_bind< VPIRValue > m_VPIRValue(VPIRValue *&V)
Match a VPIRValue.
VPInstruction_match< VPInstruction::WideActiveLaneMask, Op0_t, Op1_t, Op2_t > m_WideActiveLaneMask(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
auto m_VPPhi(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::BranchOnTwoConds > m_BranchOnTwoConds()
AllRecipe_match< Opcode, Op0_t, Op1_t > m_Binary(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::LastActiveLane, Op0_t > m_LastActiveLane(const Op0_t &Op0)
auto m_WidenIntrinsic(const T &...Ops)
canonical_widen_iv_match m_CanonicalWidenIV()
VPInstruction_match< VPInstruction::ExitingIVValue, Op0_t > m_ExitingIVValue(const Op0_t &Op0)
VPInstruction_match< Instruction::ExtractElement, Op0_t, Op1_t > m_ExtractElement(const Op0_t &Op0, const Op1_t &Op1)
specific_intval< 1 > m_False()
VPInstruction_match< VPInstruction::ExtractLastLane, Op0_t > m_ExtractLastLane(const Op0_t &Op0)
match_bind< VPSingleDefRecipe > m_VPSingleDefRecipe(VPSingleDefRecipe *&V)
Match a VPSingleDefRecipe, capturing if we match.
VPInstruction_match< VPInstruction::BranchOnCount > m_BranchOnCount()
auto m_GetElementPtr(const Op0_t &Op0, const Op1_t &Op1)
specific_intval< 1 > m_True()
auto m_VPValue()
Match an arbitrary VPValue and ignore it.
VPInstruction_match< VPInstruction::ExtractVectorForPart, Op0_t, Op1_t > m_ExtractVectorForPart(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > m_ExtractLastPart(const Op0_t &Op0)
VPRecipeBase * findUserOf(VPValue *V, const MatchT &P)
If V is used by a recipe matching pattern P, return it.
VPInstruction_match< VPInstruction::Broadcast, Op0_t > m_Broadcast(const Op0_t &Op0)
header_mask_match m_HeaderMask()
VPInstruction_match< VPInstruction::BuildVector > m_BuildVector()
BuildVector is matches only its opcode, w/o matching its operands as the number of operands is not fi...
VPInstruction_match< VPInstruction::ExtractPenultimateElement, Op0_t > m_ExtractPenultimateElement(const Op0_t &Op0)
match_bind< VPInstruction > m_VPInstruction(VPInstruction *&V)
Match a VPInstruction, capturing if we match.
VPInstruction_match< VPInstruction::FirstActiveLane, Op0_t > m_FirstActiveLane(const Op0_t &Op0)
auto m_DerivedIV(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
VPInstruction_match< VPInstruction::BranchOnCond > m_BranchOnCond()
VPInstruction_match< VPInstruction::ExtractLane, Op0_t, Op1_t > m_ExtractLane(const Op0_t &Op0, const Op1_t &Op1)
auto m_AnyNeg(const Op0_t &Op0)
VPInstruction_match< VPInstruction::Reverse, Op0_t > m_Reverse(const Op0_t &Op0)
NodeAddr< DefNode * > Def
bool isSingleScalar(const VPValue *VPV)
Returns true if VPV is a single scalar, either because it produces the same value for all lanes or on...
VPValue * getOrCreateVPValueForSCEVExpr(VPlan &Plan, const SCEV *Expr)
Get or create a VPValue that corresponds to the expansion of Expr.
bool cannotHoistOrSinkRecipe(const VPRecipeBase &R, bool Sinking=false)
Return true if we do not know how to (mechanically) hoist or sink R.
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
VPInstruction * findComputeReductionResult(VPReductionPHIRecipe *PhiR)
Find the ComputeReductionResult recipe for PhiR, looking through selects inserted for predicated redu...
VPInstruction * findCanonicalIVIncrement(VPlan &Plan)
Find the canonical IV increment of Plan's vector loop region.
std::optional< MemoryLocation > getMemoryLocation(const VPRecipeBase &R)
Return a MemoryLocation for R with noalias metadata populated from R, if the recipe is supported and ...
bool onlyFirstLaneUsed(const VPValue *Def)
Returns true if only the first lane of Def is used.
VPIRValue * tryToFoldLiveIns(VPSingleDefRecipe &R, ArrayRef< VPValue * > Operands, const DataLayout &DL)
Try to fold R using InstSimplifyFolder.
SmallVector< std::pair< VPBasicBlock *, VPIRBasicBlock * > > getEarlyExits(const VPlan &Plan, const VPBlockBase *MiddleVPBB)
Returns the (early exiting block, exit block) pairs of Plan, i.e.
void recursivelyDeleteDeadRecipes(VPValue *V)
Recursively delete V and any of its operands that become dead.
bool doesGeneratePerAllLanes(const VPRecipeBase *R)
Returns true if R produces scalar values for all VF lanes.
bool isDeadRecipe(VPRecipeBase &R)
Returns true if R is dead, i.e.
VPRecipeBase * findRecipe(VPValue *Start, PredT Pred)
Search Start's users for a recipe satisfying Pred, looking through recipes with definitions.
bool isUniformAcrossVFsAndUFs(const VPValue *V)
Checks if V is uniform across all VF lanes and UF parts.
bool isUsedByLoadStoreAddress(const VPValue *V)
Returns true if V is used as part of the address of another load or store.
std::optional< std::pair< bool, unsigned > > getOpcodeOrIntrinsicID(const VPValue *V)
Get the instruction opcode or intrinsic ID for the recipe defining V.
VPValue * scalarizeVPWidenPointerInduction(VPWidenPointerInductionRecipe *PtrIV, VPlan &Plan, VPBuilder &Builder)
Scalarize a VPWidenPointerInductionRecipe by replacing it with a PtrAdd (IndStart,...
const SCEV * getSCEVExprForVPValue(const VPValue *V, PredicatedScalarEvolution &PSE, const Loop *L=nullptr)
Return the SCEV expression for V.
void pullOutPermutations(VPlan &Plan, Match_t Perm, Builder Build)
Removes the permutation pattern Perm from any elementwise operations in the plan, by constructing a n...
SmallVector< VPUser * > collectUsersRecursively(VPValue *V)
Collect all users of V, looking through recipes that define other values.
VPScalarIVStepsRecipe * createScalarIVSteps(VPlan &Plan, InductionDescriptor::InductionKind Kind, Instruction::BinaryOps InductionOpcode, FPMathOperator *FPBinOp, Instruction *TruncI, VPIRValue *StartV, VPValue *Step, DebugLoc DL, VPBuilder &Builder, const VPIRFlags::WrapFlagsTy &Flags={})
Create a scalar-iv-steps recipe over Plan's canonical IV for an induction of Kind with InductionOpcod...
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
SmallVector< VPBasicBlock * > vp_rpo_plain_cfg_loop_body(VPBasicBlock *Header)
Returns the VPBasicBlocks forming the loop body of a plain (pre-region) VPlan in reverse post-order s...
void stable_sort(R &&Range)
auto min_element(R &&Range)
Provide wrappers to std::min_element which take ranges instead of having to pass begin/end explicitly...
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
unsigned getLoadStoreAddressSpace(const Value *I)
A helper function that returns the address space of the pointer operand of load or store instruction.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
LLVM_ABI Intrinsic::ID getVectorIntrinsicIDForCall(const CallInst *CI, const TargetLibraryInfo *TLI)
Returns intrinsic ID for call.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
DenseMap< const Value *, const SCEV * > ValueToSCEVMapTy
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr from_range_t from_range
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
auto cast_or_null(const Y &Val)
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
iterator_range< df_iterator< VPBlockShallowTraversalWrapper< VPBlockBase * > > > vp_depth_first_shallow(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order.
constexpr auto bind_back(FnT &&Fn, BindArgsT &&...BindArgs)
C++23 bind_back.
bool isa_and_nonnull(const Y &Val)
iterator_range< df_iterator< VPBlockDeepTraversalWrapper< VPBlockBase * > > > vp_depth_first_deep(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order while traversing t...
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
auto dyn_cast_or_null(const Y &Val)
void erase(Container &C, ValueType V)
Wrapper function to remove a value from a container:
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
auto reverse(ContainerTy &&C)
constexpr size_t range_size(R &&Range)
Returns the size of the Range, i.e., the number of elements.
void sort(IteratorTy Start, IteratorTy End)
DenseMap< Value *, const SCEVUnknown * > SymbolicStrideMap
Maps a pointer to its symbolic (non-constant) stride.
bool hasIrregularType(Type *Ty, const DataLayout &DL)
A helper function that returns true if the given type is irregular.
UncountableExitStyle
Different methods of handling early exits.
@ ReadOnly
No side effects to worry about, so we can process any uncountable exits in the loop and branch either...
@ MaskedHandleExitInScalarLoop
All memory operations other than the load(s) required to determine whether an uncountable exit occurr...
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
iterator_range< filter_iterator< detail::IterOfRange< RangeT >, PredicateT > > make_filter_range(RangeT &&Range, PredicateT Pred)
Convenience function that takes a range of elements and a predicate, and return a new filter_iterator...
bool canConstantBeExtended(const APInt *C, Type *NarrowType, TTI::PartialReductionExtendKind ExtKind)
Check if a constant CI can be safely treated as having been extended from a narrower type with the gi...
T * find_singleton(R &&Range, Predicate P, bool AllowRepeats=false)
Return the single value in Range that satisfies P(<member of Range> *, AllowRepeats)->T * returning n...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
RecurKind
These are the kinds of recurrences that we support.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ FindIV
FindIV reduction with select(icmp(),x,y) where one of (x,y) is a loop induction variable (increasing ...
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
@ FSub
Subtraction of floats.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
LLVM_ABI Value * getRecurrenceIdentity(RecurKind K, Type *Tp, FastMathFlags FMF)
Given information about an recurrence kind, return the identity for the @llvm.vector....
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
DWARFExpression::Operation Op
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
ArrayRef(const T &OneElt) -> ArrayRef< T >
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
LLVM_ABI std::optional< int64_t > getStrideFromAddRec(const SCEVAddRecExpr *AR, const Loop *Lp, Type *AccessTy, Value *Ptr, PredicatedScalarEvolution &PSE)
If AR is an affine AddRec for Lp with a constant step, return the step in units of AccessTy's allocat...
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI bool isDereferenceableAndAlignedInLoop(LoadInst *LI, Loop *L, ScalarEvolution &SE, DominatorTree &DT, AssumptionCache *AC=nullptr, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Return true if we can prove that the given load (which is assumed to be within the specified loop) wo...
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
hash_code hash_combine_range(InputIteratorT first, InputIteratorT last)
Compute a hash_code for a sequence of values.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
VPBasicBlock * EarlyExitingVPBB
VPIRBasicBlock * EarlyExitVPBB
This struct is a compact representation of a valid (non-zero power of two) alignment.
An information struct used to provide DenseMap with the various necessary components for a given valu...
This reduction is unordered with the partial result scaled down by some factor.
Holds the VFShape for a specific scalar to vector function mapping.
Encapsulates information needed to describe a parameter.
A range of powers-of-2 vectorization factors with fixed start and adjustable end.
Struct to hold various analysis needed for cost computations.
const VFSelectionContext & Config
static bool isFreeScalarIntrinsic(Intrinsic::ID ID)
Returns true if ID is a pseudo intrinsic that is dropped via scalarization rather than widened.
bool isMaskRequired(Instruction *I) const
Forwards to LoopVectorizationCostModel::isMaskRequired.
PredicatedScalarEvolution & PSE
bool willBeScalarized(Instruction *I, ElementCount VF) const
Returns true if I is known to be scalarized at VF.
TargetTransformInfo::TargetCostKind CostKind
const TargetLibraryInfo & TLI
const TargetTransformInfo & TTI
A VPValue representing a live-in from the input IR or a constant.
Type * getType() const
Returns the type of the underlying IR value.
A recipe for widening load operations, using the address to load from and an optional mask.
A recipe for widening store operations, using the stored value, the address to store to and an option...